From 047e74d507e8153340c52df7a135e3a1fbd53667 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Mon, 5 Oct 2026 14:00:51 +0000 Subject: [PATCH] docs: the Redux trim regression was noise, keep the 0.3 default The first trim run (PR 94) read a small Redux loss on talks and on pink noise at 0 dB. It was repeated on 9 talks and about 375 noisy files with paired bootstrap intervals: no interval for talks or noisy speech excludes zero for the Ultra head, the Redux head or Silero with TDT v3. The earlier sets were too small (one word is 0.23 points) and Redux changes its text for tiny pad shifts. Correct docs/vad-benchmarks.md and docs/vad.md, keep the first numbers labelled as superseded, add the follow-up tables, and commit the compact scripts and results. No code changes. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] --- docs/vad-benchmarks.md | 78 ++++++++- docs/vad.md | 10 +- scripts/vad_bench/decoder_guards/README.md | 6 + .../results/trim_regression_followup.md | 88 +++++++++++ scripts/vad_bench/trim_regression/README.md | 48 ++++++ .../vad_bench/trim_regression/make_results.sh | 23 +++ .../trim_regression/results/audio_clip.txt | 27 ++++ .../results/boundary_T0_T0.3.txt | 36 +++++ .../results/cand_redux_heldout.md | 17 ++ .../results/cand_redux_tune.md | 17 ++ .../results/cand_ultra_heldout.md | 9 ++ .../results/cand_ultra_tune.md | 9 ++ .../results/cand_v3_heldout.md | 9 ++ .../trim_regression/results/cand_v3_tune.md | 9 ++ .../trim_regression/results/clip_count.txt | 66 ++++++++ .../trim_regression/results/grid_redux_all.md | 10 ++ .../results/grid_redux_heldout.md | 10 ++ .../results/grid_redux_tune.md | 10 ++ .../results/grid_ultra_talks_all.txt | 1 + .../results/grid_v3_talks_all.txt | 1 + .../results/noise_seconds_90_inserts.txt | 35 ++++ .../results/noise_words_20_inserts.md | 21 +++ .../results/per_detector_default_table.md | 19 +++ .../trim_regression/results/pooled.md | 16 ++ .../trim_regression/results/removed.txt | 9 ++ .../vad_bench/trim_regression/results/sdi.txt | 7 + .../trim_regression/results/segchange.txt | 14 ++ .../trim_regression/results/vad_cover.txt | 12 ++ .../trim_regression/results/vad_edges.txt | 149 ++++++++++++++++++ .../results/validate_cli_vs_harness.txt | 33 ++++ .../trim_regression/scripts/audio_clip.py | 21 +++ .../trim_regression/scripts/boundary.py | 57 +++++++ .../trim_regression/scripts/clip_count.py | 27 ++++ .../trim_regression/scripts/common.py | 19 +++ .../vad_bench/trim_regression/scripts/dec.py | 34 ++++ .../trim_regression/scripts/fetch.py | 26 +++ .../vad_bench/trim_regression/scripts/gens.py | 29 ++++ .../trim_regression/scripts/groups.py | 27 ++++ .../vad_bench/trim_regression/scripts/lib.py | 90 +++++++++++ .../trim_regression/scripts/mkcorpus.py | 69 ++++++++ .../trim_regression/scripts/noise_ins.py | 18 +++ .../trim_regression/scripts/noise_words.py | 22 +++ .../trim_regression/scripts/pooled.py | 18 +++ .../trim_regression/scripts/probe.py | 19 +++ .../trim_regression/scripts/ready.py | 10 ++ .../trim_regression/scripts/removed.py | 15 ++ .../trim_regression/scripts/report_cand.py | 30 ++++ .../trim_regression/scripts/report_grid.py | 22 +++ .../vad_bench/trim_regression/scripts/run.py | 33 ++++ .../vad_bench/trim_regression/scripts/seg.py | 73 +++++++++ .../trim_regression/scripts/segchange.py | 27 ++++ .../trim_regression/scripts/table_trim.py | 22 +++ .../trim_regression/scripts/vad_cover.py | 30 ++++ .../trim_regression/scripts/vad_edges.py | 50 ++++++ .../trim_regression/scripts/validate.py | 26 +++ .../trim_regression/scripts/variants.py | 14 ++ .../trim_regression/throwaway_cli_patch.diff | 50 ++++++ 57 files changed, 1670 insertions(+), 7 deletions(-) create mode 100644 scripts/vad_bench/decoder_guards/results/trim_regression_followup.md create mode 100644 scripts/vad_bench/trim_regression/README.md create mode 100755 scripts/vad_bench/trim_regression/make_results.sh create mode 100644 scripts/vad_bench/trim_regression/results/audio_clip.txt create mode 100644 scripts/vad_bench/trim_regression/results/boundary_T0_T0.3.txt create mode 100644 scripts/vad_bench/trim_regression/results/cand_redux_heldout.md create mode 100644 scripts/vad_bench/trim_regression/results/cand_redux_tune.md create mode 100644 scripts/vad_bench/trim_regression/results/cand_ultra_heldout.md create mode 100644 scripts/vad_bench/trim_regression/results/cand_ultra_tune.md create mode 100644 scripts/vad_bench/trim_regression/results/cand_v3_heldout.md create mode 100644 scripts/vad_bench/trim_regression/results/cand_v3_tune.md create mode 100644 scripts/vad_bench/trim_regression/results/clip_count.txt create mode 100644 scripts/vad_bench/trim_regression/results/grid_redux_all.md create mode 100644 scripts/vad_bench/trim_regression/results/grid_redux_heldout.md create mode 100644 scripts/vad_bench/trim_regression/results/grid_redux_tune.md create mode 100644 scripts/vad_bench/trim_regression/results/grid_ultra_talks_all.txt create mode 100644 scripts/vad_bench/trim_regression/results/grid_v3_talks_all.txt create mode 100644 scripts/vad_bench/trim_regression/results/noise_seconds_90_inserts.txt create mode 100644 scripts/vad_bench/trim_regression/results/noise_words_20_inserts.md create mode 100644 scripts/vad_bench/trim_regression/results/per_detector_default_table.md create mode 100644 scripts/vad_bench/trim_regression/results/pooled.md create mode 100644 scripts/vad_bench/trim_regression/results/removed.txt create mode 100644 scripts/vad_bench/trim_regression/results/sdi.txt create mode 100644 scripts/vad_bench/trim_regression/results/segchange.txt create mode 100644 scripts/vad_bench/trim_regression/results/vad_cover.txt create mode 100644 scripts/vad_bench/trim_regression/results/vad_edges.txt create mode 100644 scripts/vad_bench/trim_regression/results/validate_cli_vs_harness.txt create mode 100644 scripts/vad_bench/trim_regression/scripts/audio_clip.py create mode 100644 scripts/vad_bench/trim_regression/scripts/boundary.py create mode 100644 scripts/vad_bench/trim_regression/scripts/clip_count.py create mode 100644 scripts/vad_bench/trim_regression/scripts/common.py create mode 100644 scripts/vad_bench/trim_regression/scripts/dec.py create mode 100644 scripts/vad_bench/trim_regression/scripts/fetch.py create mode 100644 scripts/vad_bench/trim_regression/scripts/gens.py create mode 100644 scripts/vad_bench/trim_regression/scripts/groups.py create mode 100644 scripts/vad_bench/trim_regression/scripts/lib.py create mode 100644 scripts/vad_bench/trim_regression/scripts/mkcorpus.py create mode 100644 scripts/vad_bench/trim_regression/scripts/noise_ins.py create mode 100644 scripts/vad_bench/trim_regression/scripts/noise_words.py create mode 100644 scripts/vad_bench/trim_regression/scripts/pooled.py create mode 100644 scripts/vad_bench/trim_regression/scripts/probe.py create mode 100644 scripts/vad_bench/trim_regression/scripts/ready.py create mode 100644 scripts/vad_bench/trim_regression/scripts/removed.py create mode 100644 scripts/vad_bench/trim_regression/scripts/report_cand.py create mode 100644 scripts/vad_bench/trim_regression/scripts/report_grid.py create mode 100644 scripts/vad_bench/trim_regression/scripts/run.py create mode 100644 scripts/vad_bench/trim_regression/scripts/seg.py create mode 100644 scripts/vad_bench/trim_regression/scripts/segchange.py create mode 100644 scripts/vad_bench/trim_regression/scripts/table_trim.py create mode 100644 scripts/vad_bench/trim_regression/scripts/vad_cover.py create mode 100644 scripts/vad_bench/trim_regression/scripts/vad_edges.py create mode 100644 scripts/vad_bench/trim_regression/scripts/validate.py create mode 100644 scripts/vad_bench/trim_regression/scripts/variants.py create mode 100644 scripts/vad_bench/trim_regression/throwaway_cli_patch.diff diff --git a/docs/vad-benchmarks.md b/docs/vad-benchmarks.md index 059e52f..71da631 100644 --- a/docs/vad-benchmarks.md +++ b/docs/vad-benchmarks.md @@ -952,7 +952,76 @@ claimed. "Seconds decoded" is the overlap of the cuts with the block, from `vad --mode segments`. - Noise alone: 63 files of 30 s (seven noise types, three levels), decoded whole. -### Trimming: word error rate +### Trimming: word error rate, enlarged measurement + +The first run (below) had small sets and read a small loss for the Redux head as real. It was +repeated on a larger set, with the same trim of 0.3 s against the old cuts (`--vad-trim 0`): + +- 9 whole TED-LIUM talks (21,540 reference words), 4 used to tune and 5 held out. +- About 375 noisy LibriSpeech files (6 utterances with gaps each, white and pink noise at 20, 10, 5 + and 0 dB SNR, 3 noise seeds, split by speaker into tune and held out). The rows below use white 5 + dB, pink 5 dB and pink 0 dB (15,642 reference words for Ultra and v3, 17,946 for Redux). +- Models: Redux packed, Ultra Q8_0, and TDT 0.6B v3 Q8_0 with Silero F16. The first run used F16 + files, so the two runs differ in the models as well as in the size. +- The delta is trim 0.3 minus the old cuts, in WER points (negative means trim is better), with a + 95 percent interval from a paired bootstrap over talks and utterances. + +| Set | Redux head | Ultra head | v3 + Silero | +| --- | ---: | ---: | ---: | +| 9 talks | +0.08 (-0.03..+0.20) | +0.01 (-0.06..+0.08) | +0.01 (-0.06..+0.10) | +| 5 held-out talks | +0.16 (-0.01..+0.36) | +0.02 (-0.06..+0.11) | +0.03 (-0.08..+0.20) | +| 4 tune talks | -0.01 (-0.14..+0.13) | +0.00 (-0.10..+0.12) | -0.02 (-0.09..+0.04) | +| noisy speech (white 5, pink 5, pink 0) | -0.12 (-0.57..+0.45) | -0.27 (-0.57..+0.01) | -0.26 (-0.57..+0.02) | +| pink 0 dB alone (5,982 words) | -0.15 (-0.90..+0.71) | | | + +No interval for talks or for noisy speech excludes zero. The Redux loss of the first run (4.39 to +4.51 on talks, 12.35 to 13.29 on pink 0 dB) does not hold up: on 9 talks it is +0.08, and on pink 0 +dB it has the other sign. The first run looked worse for two reasons. It was small: its four noisy +sets had 429 words each, so one word is 0.23 points and the +0.93 was four words. And Redux is the +most volatile of the three: trim 0.3 changes the text of 15.5 percent of its talk segments, against +10.0 percent for Ultra and 7.6 percent for v3, and moving the pad from 0.30 to 0.32 s (a change that +cannot matter) changes the text of 13.5 percent of them and the talk WER by -0.05 (0.35 s: -0.07, +16.6 percent). That is the noise floor of this comparison. Of the +18 net errors that Redux gains on +talks, none are at the cut edges (0 net) and 18 are in the interior of the segments, where the +decoder sees a slightly different input; only 1 of 21,459 words is cut away. + +Redux with other trims, against the old cuts (talks / noisy speech with white and pink noise at +four levels, 47,856 words), is not monotonic in the trim: 0.1 gives +0.03 / -0.01, 0.2 gives +0.02 / +-0.11, 0.3 gives +0.08 / -0.12, 0.5 gives -0.02 / +0.06, and 1.0 gives -0.00 / -0.01. Three other +rules (0.5 s before and after, 0.5 s before and 0.3 s after, and trimming only the edges of at least +0.5 s) are within the intervals of 0.3 on every set, and none is better than the default. Padding is +nearly free in noise: with trim 1.0 the decoder still gets only 5.7 s (Redux), 13.3 s (Ultra) and +0.6 s (v3) of a 60 s noise block, against 4.6, 12.1 and 0.0 s with 0.3 (old cuts: 33.2, 37.2, 27.0). +The 0.3 pad cuts away few words: on noisy speech 14 correct words of 47.8k (Redux), 14 of 16.1k +(Ultra) and 19 of 15.5k (v3), counted as words whose time lies outside the kept audio (a lower +bound). The median lateness of the detected speech start against the true start is -5 ms for +Redux, 79 ms for Ultra and 237 ms for Silero. + +Trim also has a real benefit that the first run could not see. With Silero the old cuts drop whole +sentences on clean speech when a long segment starts with silence: on the held-out clean files the v3 +WER goes from 6.37 to 3.19 with trim 0.3 (-3.19, interval -7.76..-0.14). That interval and the +Redux clean interval (-0.37, -0.80..-0.07 on held-out clean) are the only ones at 0.3 that exclude +zero, and both favour the trim. + +Decision: the default stays 0.3 s. No code or option changed with this measurement; a minimum edge +length for trimming (`trim_min_sec`) was considered and not added. + +Limits: + +- The noise is synthetic and added to read speech. Real room noise and overlapping speech were not + tested. +- The models are not the ones of the first run (see above). +- The truth for where speech starts and ends is loose: it is the span of each utterance in the + synthetic file, so a pause inside an utterance counts as speech. +- Only 5 talks are held out, and the clean sets are small (1,350 words held out). +- There is no correction for the many comparisons in the tables. Read a single interval as a + range, not as a test. + +Tables, the scripts and a note on how to regenerate them: +[scripts/vad_bench/trim_regression](../scripts/vad_bench/trim_regression/README.md) and +[results/trim_regression_followup.md](../scripts/vad_bench/decoder_guards/results/trim_regression_followup.md). + +### Trimming: word error rate, first run (small sets, superseded) | Set | Detector | Old | Trim 0.3 | Change | | --- | --- | ---: | ---: | ---: | @@ -963,9 +1032,10 @@ claimed. | white noise 5 dB | Ultra / Redux / v3 + Silero | 4.90 / 7.23 / 5.59 | 4.43 / 6.76 / 5.13 | -0.47 / -0.47 / -0.47 | | pink noise 0 dB | Ultra / Redux / v3 + Silero | 6.53 / 12.35 / 7.93 | 5.59 / 13.29 / 7.69 | -0.93 / +0.93 / -0.23 | -On talks the change is within 0.12 points; the Redux head loses a little on two of the three talks -and on pink noise. The speech-in-noise sets are small (one word is 0.23 points), so read them as -"neutral", not as a gain. +These are the numbers of the first run, kept for the record. The text that went with them said the +Redux head loses a little on talks and on pink noise. The enlarged measurement above shows that +this was noise: the sets are too small to tell (one word is 0.23 points), so read them as +"neutral". ### Trimming: the noise block diff --git a/docs/vad.md b/docs/vad.md index c7da89d..3fb7a1a 100644 --- a/docs/vad.md +++ b/docs/vad.md @@ -124,9 +124,13 @@ whole file. `trim` 0 (`--vad-trim 0`) gives the previous cuts exactly. This is a change of default behaviour for `transcribe --vad`, `parakeet_capi_transcribe_path_json_vad*` and the `segments` mode of the VAD functions, for the head, for Silero and for VAD-only slices (they share the -segmenter). On talks, transcripts of long audio can shift slightly (a word WER -cost of about 0.1 point in our runs); on audio with long noisy stretches the -decoder sees much less noise. Numbers: [vad-benchmarks.md](vad-benchmarks.md#trimming-segments-and-the-word-filter). +segmenter). Transcripts of long audio can shift slightly. An enlarged measurement +(9 talks and about 375 noisy files) found no change in word error rate whose +interval excludes zero, for the Ultra head, the Redux head or Silero with TDT v3; +an earlier small run that read a loss for the Redux head was noise. On audio with +long noisy stretches the decoder sees much less noise, and with Silero the trim +stops whole sentences from being dropped on clean speech. Numbers: +[vad-benchmarks.md](vad-benchmarks.md#trimming-segments-and-the-word-filter). ## Word filter (opt-in) diff --git a/scripts/vad_bench/decoder_guards/README.md b/scripts/vad_bench/decoder_guards/README.md index 9fa8198..79086ac 100644 --- a/scripts/vad_bench/decoder_guards/README.md +++ b/scripts/vad_bench/decoder_guards/README.md @@ -36,3 +36,9 @@ What the files are: `--vad-trim 0` gives the old output byte for byte, the WER with the old cuts, with trim 0.3 and with trim 0.3 plus `--min-local-conf 0.5`, the seconds of the noise block that the decoder gets, and the words it returns inside the block. + +Note on the trim results: `results/tables.txt` is the output of the first run, which has small +sets. Its Redux rows (talks 4.39 to 4.51, pink noise 0 dB 12.35 to 13.29) read as a small loss for +the Redux head. An enlarged measurement showed that this was noise (no interval excludes zero); the +file is kept unchanged. See `results/trim_regression_followup.md` and +[../trim_regression](../trim_regression/README.md). diff --git a/scripts/vad_bench/decoder_guards/results/trim_regression_followup.md b/scripts/vad_bench/decoder_guards/results/trim_regression_followup.md new file mode 100644 index 0000000..daae135 --- /dev/null +++ b/scripts/vad_bench/decoder_guards/results/trim_regression_followup.md @@ -0,0 +1,88 @@ +# Trim regression follow-up: tables + +Tables behind the section "Trimming: word error rate, enlarged measurement" of +[docs/vad-benchmarks.md](../../../../docs/vad-benchmarks.md). Every table is copied from a file in +[../../trim_regression/results](../../trim_regression/results) (named in each heading); the scripts +and the way to regenerate them are in [../../trim_regression](../../trim_regression/README.md). +`results/tables.txt` in this directory is the first run (small sets) and is not changed. Its Redux +rows (talks 4.39 to 4.51, pink noise 0 dB 12.35 to 13.29) were noise. + +Deltas are trim 0.3 minus old cuts (`--vad-trim 0`) in WER points, with a 95 percent paired +bootstrap interval. T0 is the old cuts, Tx is trim x seconds. + +## Trim 0.3 against the old cuts (sdi.txt, grid_redux_*.md, cand_*_heldout.md, cand_*_tune.md, grid_*_talks_all.txt) + +| Set | Redux head | Ultra head | v3 + Silero | +| --- | ---: | ---: | ---: | +| 9 talks (21,540 words) | +0.08 (-0.03..+0.20) | +0.01 (-0.06..+0.08) | +0.01 (-0.06..+0.10) | +| 5 held-out talks (11,490 words) | +0.16 (-0.01..+0.36) | +0.02 (-0.06..+0.11) | +0.03 (-0.08..+0.20) | +| 4 tune talks (10,050 words) | -0.01 (-0.14..+0.13) | +0.00 (-0.10..+0.12) | -0.02 (-0.09..+0.04) | +| noisy speech: white 5, pink 5, pink 0 (Redux 17,946 words, others 15,642) | -0.12 (-0.57..+0.45) | -0.27 (-0.57..+0.01) | -0.26 (-0.57..+0.02) | +| pink 0 dB alone, Redux (5,982 words) | -0.15 (-0.90..+0.71) | | | +| clean speech, held out (1,350 words) | -0.37 (-0.80..-0.07) | -0.22 (-0.76..+0.28) | -3.19 (-7.76..-0.14) | + +Held-out clean WER for v3 + Silero: 6.37 with the old cuts, 3.19 with trim 0.3. + +## Redux, trim variants against the old cuts (grid_redux_all.md, 9 talks and all noisy files) + +noisy-all is white and pink noise at 20, 10, 5 and 0 dB (47,856 words). + +| Variant | talks | noisy-all | +| --- | ---: | ---: | +| T0.1 | +0.03 (-0.12..+0.18) | -0.01 (-0.22..+0.22) | +| T0.2 | +0.02 (-0.13..+0.19) | -0.11 (-0.40..+0.16) | +| T0.3 | +0.08 (-0.03..+0.20) | -0.12 (-0.42..+0.17) | +| T0.5 | -0.02 (-0.08..+0.04) | +0.06 (-0.13..+0.25) | +| T1.0 | -0.00 (-0.01..+0.00) | -0.01 (-0.13..+0.12) | + +## Text changes when the trim changes (segchange.txt) + +Share of segments whose text changes, and the net change in errors. + +| Detector | T0 to T0.3, talks | T0 to T0.3, noisy | T0.3 to T0.5, talks | +| --- | ---: | ---: | ---: | +| Redux | 46 of 296 (15.5%), net +17 | 439 of 860 (51.0%), net -56 | 49 of 296 (16.6%), net -22 | +| Ultra | 30 of 299 (10.0%), net +1 | 149 of 291 (51.2%), net -40 | 29 of 299 (9.7%), net +0 | +| v3 + Silero | 21 of 276 (7.6%), net +2 | 151 of 279 (54.1%), net -32 | 20 of 276 (7.2%), net -3 | + +Redux, pad 0.30 to 0.32 s: 40 of 296 segments (13.5%) change text; 0.30 to 0.35 s: 49 of 296 +(16.6%). On the 9 talks (weighted from the tune and held-out sets of cand_redux_*.md) the talk WER +is 4.87 with 0.30, 4.82 with 0.32 and 4.80 with 0.35, that is -0.05 and -0.07. + +## Where the errors move, Redux, T0 to T0.3 (boundary_T0_T0.3.txt) + +Errors by zone of the segment (substitutions, deletions, insertions). + +| Set | cut away | start edge | end edge | interior | +| --- | ---: | ---: | ---: | ---: | +| talks | 2 to 1 | 18 to 18 | 21 to 21 | 992 to 1010 (+18) | +| noisy speech | 4 to 5 | 21 to 15 | 31 to 13 | 3208 to 3169 (-39) | + +Words cut away on noisy speech that were correct, by trim 0.3 (clip_count.txt): Redux 14 of 47,791, +Ultra 14 of 16,133, v3 19 of 15,506. + +## Noise block (noise_seconds_90_inserts.txt, noise_words_20_inserts.md) + +Seconds of a 60 s noise block that the decoder gets (90 insert files), and invented words in the +block (20 insert files). + +| Detector | T0 | T0.3 | T0.5 | T1.0 | words, T0 | words, T0.3 | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| Redux | 33.2 | 4.6 | 4.9 | 5.7 | 0 | 0 | +| Ultra | 37.2 | 12.1 | 12.4 | 13.3 | 18 | 0 | +| v3 + Silero | 27.0 | 0.0 | 0.2 | 0.6 | 14 | 0 | + +## Detected speech start against the true start (vad_edges.txt, all noisy files) + +Median lateness: Redux -5 ms, Ultra 79 ms, Silero 237 ms. + +## Other padding rules (per_detector_default_table.md) + +Held-out WER (talks, clean and the three noisy conditions together) and the delta against T0.3: +no rule is better than 0.3 outside the intervals. + +| Detector | no trim | 0.3 / 0.3 | 0.5 / 0.5 | 0.5 / 0.3 | 0.3 / 0.3, edges of at least 0.5 s only | +| --- | ---: | ---: | ---: | ---: | ---: | +| Redux | 6.34 | 6.40 | 6.32 | 6.33 | 6.31 | +| Ultra | 5.10 | 5.01 | 5.05 | 5.05 | 5.04 | +| v3 + Silero | 5.70 | 5.40 | 5.52 | 5.51 | 5.41 | diff --git a/scripts/vad_bench/trim_regression/README.md b/scripts/vad_bench/trim_regression/README.md new file mode 100644 index 0000000..c303345 --- /dev/null +++ b/scripts/vad_bench/trim_regression/README.md @@ -0,0 +1,48 @@ +# Trim regression follow-up + +Scripts and result files behind the section "Trimming: word error rate, enlarged measurement" of +[docs/vad-benchmarks.md](../../../docs/vad-benchmarks.md). The first run of the trim change (see +[../decoder_guards](../decoder_guards/README.md)) read a small loss for the Redux head. This +follow-up repeated the comparison on 9 talks and about 375 noisy files with paired bootstrap +intervals and found that the loss was noise. + +No audio, no model files, no probability files and no decode cache are committed. The tables that +the page quotes are in [../decoder_guards/results/trim_regression_followup.md](../decoder_guards/results/trim_regression_followup.md). + +## What is here + +| Path | What | +| --- | --- | +| `scripts/` | Python scripts. `common.py` holds the paths and the segmenter settings, `seg.py` a Python port of the segmenter (with pad before, pad after, minimum edge and minimum length options), `dec.py` the segment decode cache, `lib.py` the scoring and the paired bootstrap | +| `make_results.sh` | Rebuilds the tables in `results/` from the decode cache; it runs no ASR | +| `throwaway_cli_patch.diff` | A patch of `src/model.cpp` for measurement only. It is not part of the product. It adds three environment variables to `parakeet-cli` (dump the VAD probabilities, decode a given list of segments, print the words of each slice) | +| `results/` | The output of `make_results.sh`, plus `validate_cli_vs_harness.txt` | + +Some tables (`results/sdi.txt`, `results/per_detector_default_table.md`) were made by one-off +commands on the same cache; their scripts were not kept, so `make_results.sh` does not rebuild +them. The other files in `results/` are rebuilt by it. + +## How to regenerate + +Environment variables: `PK_WORK` is the work directory (default: the current directory; it holds +`data/`, `probs/`, `cache/` and `tmp/`), `PK_CLI` is the patched `parakeet-cli` +(default `$PK_WORK/build/examples/cli/parakeet-cli`), `PK_GGUF` is a directory with +`ultra-q8_0.gguf`, `redux-keep.gguf` (the Redux head, packed), `tdt-0.6b-v3-q8_0.gguf` and +`silero-vad-f16.gguf`. Python packages: `numpy soundfile jiwer librosa datasets`. + +``` +git apply throwaway_cli_patch.diff # in a scratch checkout; build parakeet-cli from it +export PK_WORK=/path/to/work PK_CLI=/path/to/patched/parakeet-cli PK_GGUF=/path/to/models +python3 scripts/fetch.py $PK_WORK/data # TED-LIUM long-form talks and LibriSpeech test-clean (streamed) +python3 scripts/mkcorpus.py $PK_WORK/data # speech in noise sets, noise inserts, manifest.json (fixed seeds) +python3 scripts/probe.py 4 # VAD probabilities of every file, per detector +python3 scripts/run.py T0,T0.1,T0.2,T0.3,T0.5,T1.0 talk,sinr,insert # decode the segments of each variant (hours of CPU, resumes) +python3 scripts/validate.py # the harness against the patched CLI: segment bounds and text must match +sh make_results.sh # tables into $PK_WORK/results +rm -r $PK_WORK/data # audio +``` + +`scripts/variants.py` defines the variants: `Tx` is trim x seconds, `T0` the old cuts, `Px/y` a +pad of x before and y after, `Ex` trim only edges of at least x seconds, `Lx` trim only segments of +at least x seconds. `results/validate_cli_vs_harness.txt` shows 30 of 30 runs with the same segment +bounds and the same text as the CLI. diff --git a/scripts/vad_bench/trim_regression/make_results.sh b/scripts/vad_bench/trim_regression/make_results.sh new file mode 100755 index 0000000..0cc23fa --- /dev/null +++ b/scripts/vad_bench/trim_regression/make_results.sh @@ -0,0 +1,23 @@ +#!/bin/sh +# Regenerates every table in results/ from the decode cache (no ASR run). +# Run it from the work directory (or set PK_WORK); the scripts are next to this file. +# PYTHON is the interpreter (needs numpy, soundfile, jiwer). +HERE="$(cd "$(dirname "$0")" && pwd)" +export PK_WORK="${PK_WORK:-$(pwd)}" +P="${PYTHON:-python3}"; S="$HERE/scripts"; O="$PK_WORK/results"; mkdir -p "$O" +$P $S/report_grid.py redux all T0,T0.1,T0.2,T0.3,T0.5,T1.0 > $O/grid_redux_all.md +for s in tune heldout; do $P $S/report_grid.py redux $s T0,T0.1,T0.2,T0.3,T0.5,T1.0 > $O/grid_redux_$s.md; done +for d in ultra v3; do for s in tune heldout; do $P $S/report_cand.py $d $s T0,T0.3,T0.5,P0.5/0.3,E0.5 T0 white5,pink5,pink0 > $O/cand_${d}_$s.md; done; done +for s in tune heldout; do $P $S/report_cand.py redux $s T0,T0.1,T0.2,T0.3,T0.32,T0.35,T0.5,T1.0,P0.5/0.3,E0.5,E1.0,E0.5p0.5,L3 T0 white5,white0,pink5,pink0 > $O/cand_redux_$s.md; done +for d in ultra v3; do $P $S/table_trim.py T0,T0.1,T0.2,T0.3,T0.5,T1.0 all $d | grep talks > $O/grid_${d}_talks_all.txt; done +for d in redux ultra v3; do for k in talk sinr; do $P $S/boundary.py $d T0 T0.3 all $k; done; done > $O/boundary_T0_T0.3.txt +for d in redux ultra v3; do for p in "T0 T0.3" "T0.3 T0.5"; do $P $S/segchange.py $d $p talk,sinr; done; done > $O/segchange.txt 2>&1 +$P $S/segchange.py redux T0.3 T0.32 talk >> $O/segchange.txt; $P $S/segchange.py redux T0.3 T0.35 talk >> $O/segchange.txt +$P $S/clip_count.py T0.1,T0.2,T0.3,T0.5,T1.0,P0.5/0.3 > $O/clip_count.txt 2>&1 +$P $S/audio_clip.py T0.1,T0.2,T0.3,T0.4,T0.5,T1.0,P0.5/0.3,P0.5/0.4 ultra,redux,v3 > $O/audio_clip.txt 2>&1 +$P $S/vad_edges.py > $O/vad_edges.txt 2>&1 +$P $S/vad_cover.py > $O/vad_cover.txt 2>&1 +$P $S/removed.py > $O/removed.txt 2>&1 +$P $S/noise_ins.py T0,T0.1,T0.3,T0.5,T1.0,P0.5/0.3,E0.5,E1.0,E0.5p0.5,L3,L6 > $O/noise_seconds_90_inserts.txt 2>&1 +$P $S/noise_words.py T0,T0.3,T0.5,P0.5/0.3,E0.5,L3 redux,ultra,v3 > $O/noise_words_20_inserts.md 2>&1 +$P $S/pooled.py > $O/pooled.md diff --git a/scripts/vad_bench/trim_regression/results/audio_clip.txt b/scripts/vad_bench/trim_regression/results/audio_clip.txt new file mode 100644 index 0000000..afb2ad4 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/audio_clip.txt @@ -0,0 +1,27 @@ +== ultra + T0.1 segment starts: cut into the first utterance 64.2% (>=50ms 58.6%, >=100ms 54.1%, >=300ms 22.0%, max 641ms) | ends: cut into the last utterance 51.0% (>=100ms 40.1%, >=300ms 12.4%, max 665ms) n=573 + T0.2 segment starts: cut into the first utterance 54.1% (>=50ms 47.5%, >=100ms 42.4%, >=300ms 7.3%, max 541ms) | ends: cut into the last utterance 40.1% (>=100ms 22.3%, >=300ms 5.8%, max 565ms) n=573 + T0.3 segment starts: cut into the first utterance 42.4% (>=50ms 28.4%, >=100ms 22.0%, >=300ms 1.2%, max 441ms) | ends: cut into the last utterance 22.3% (>=100ms 12.4%, >=300ms 2.6%, max 465ms) n=573 + T0.4 segment starts: cut into the first utterance 22.0% (>=50ms 18.0%, >=100ms 7.3%, >=300ms 0.2%, max 341ms) | ends: cut into the last utterance 12.4% (>=100ms 5.8%, >=300ms 1.7%, max 365ms) n=573 + T0.5 segment starts: cut into the first utterance 7.3% (>=50ms 4.0%, >=100ms 1.2%, >=300ms 0.0%, max 241ms) | ends: cut into the last utterance 5.8% (>=100ms 2.6%, >=300ms 0.0%, max 265ms) n=573 + T1.0 segment starts: cut into the first utterance 0.0% (>=50ms 0.0%, >=100ms 0.0%, >=300ms 0.0%, max 0ms) | ends: cut into the last utterance 0.0% (>=100ms 0.0%, >=300ms 0.0%, max 0ms) n=573 + P0.5/0.3 segment starts: cut into the first utterance 7.3% (>=50ms 4.0%, >=100ms 1.2%, >=300ms 0.0%, max 241ms) | ends: cut into the last utterance 22.3% (>=100ms 12.4%, >=300ms 2.6%, max 465ms) n=573 + P0.5/0.4 segment starts: cut into the first utterance 7.3% (>=50ms 4.0%, >=100ms 1.2%, >=300ms 0.0%, max 241ms) | ends: cut into the last utterance 12.4% (>=100ms 5.8%, >=300ms 1.7%, max 365ms) n=573 +== redux + T0.1 segment starts: cut into the first utterance 55.9% (>=50ms 55.2%, >=100ms 52.3%, >=300ms 18.0%, max 481ms) | ends: cut into the last utterance 40.4% (>=100ms 29.2%, >=300ms 4.7%, max 585ms) n=596 + T0.2 segment starts: cut into the first utterance 52.3% (>=50ms 42.6%, >=100ms 33.4%, >=300ms 4.0%, max 381ms) | ends: cut into the last utterance 29.2% (>=100ms 8.7%, >=300ms 2.0%, max 485ms) n=596 + T0.3 segment starts: cut into the first utterance 33.4% (>=50ms 21.5%, >=100ms 18.0%, >=300ms 0.0%, max 281ms) | ends: cut into the last utterance 8.7% (>=100ms 4.7%, >=300ms 0.5%, max 385ms) n=596 + T0.4 segment starts: cut into the first utterance 18.0% (>=50ms 12.2%, >=100ms 4.0%, >=300ms 0.0%, max 181ms) | ends: cut into the last utterance 4.7% (>=100ms 2.0%, >=300ms 0.0%, max 285ms) n=596 + T0.5 segment starts: cut into the first utterance 4.0% (>=50ms 2.2%, >=100ms 0.0%, >=300ms 0.0%, max 81ms) | ends: cut into the last utterance 2.0% (>=100ms 0.5%, >=300ms 0.0%, max 185ms) n=596 + T1.0 segment starts: cut into the first utterance 0.0% (>=50ms 0.0%, >=100ms 0.0%, >=300ms 0.0%, max 0ms) | ends: cut into the last utterance 0.0% (>=100ms 0.0%, >=300ms 0.0%, max 0ms) n=596 + P0.5/0.3 segment starts: cut into the first utterance 4.0% (>=50ms 2.2%, >=100ms 0.0%, >=300ms 0.0%, max 81ms) | ends: cut into the last utterance 8.7% (>=100ms 4.7%, >=300ms 0.5%, max 385ms) n=596 + P0.5/0.4 segment starts: cut into the first utterance 4.0% (>=50ms 2.2%, >=100ms 0.0%, >=300ms 0.0%, max 81ms) | ends: cut into the last utterance 4.7% (>=100ms 2.0%, >=300ms 0.0%, max 285ms) n=596 +== v3 + T0.1 segment starts: cut into the first utterance 66.1% (>=50ms 65.7%, >=100ms 65.5%, >=300ms 33.1%, max 1521ms) | ends: cut into the last utterance 55.1% (>=100ms 46.1%, >=300ms 12.2%, max 788ms) n=499 + T0.2 segment starts: cut into the first utterance 65.5% (>=50ms 52.9%, >=100ms 50.1%, >=300ms 15.0%, max 1421ms) | ends: cut into the last utterance 46.1% (>=100ms 30.1%, >=300ms 9.0%, max 688ms) n=499 + T0.3 segment starts: cut into the first utterance 50.1% (>=50ms 35.7%, >=100ms 33.1%, >=300ms 4.8%, max 1321ms) | ends: cut into the last utterance 30.1% (>=100ms 12.2%, >=300ms 2.4%, max 588ms) n=499 + T0.4 segment starts: cut into the first utterance 33.1% (>=50ms 26.1%, >=100ms 15.0%, >=300ms 1.0%, max 1221ms) | ends: cut into the last utterance 12.2% (>=100ms 9.0%, >=300ms 0.6%, max 488ms) n=499 + T0.5 segment starts: cut into the first utterance 15.0% (>=50ms 11.2%, >=100ms 4.8%, >=300ms 0.6%, max 1121ms) | ends: cut into the last utterance 9.0% (>=100ms 2.4%, >=300ms 0.4%, max 388ms) n=499 + T1.0 segment starts: cut into the first utterance 0.6% (>=50ms 0.6%, >=100ms 0.6%, >=300ms 0.6%, max 621ms) | ends: cut into the last utterance 0.0% (>=100ms 0.0%, >=300ms 0.0%, max 0ms) n=499 + P0.5/0.3 segment starts: cut into the first utterance 15.0% (>=50ms 11.2%, >=100ms 4.8%, >=300ms 0.6%, max 1121ms) | ends: cut into the last utterance 30.1% (>=100ms 12.2%, >=300ms 2.4%, max 588ms) n=499 + P0.5/0.4 segment starts: cut into the first utterance 15.0% (>=50ms 11.2%, >=100ms 4.8%, >=300ms 0.6%, max 1121ms) | ends: cut into the last utterance 12.2% (>=100ms 9.0%, >=300ms 0.6%, max 488ms) n=499 diff --git a/scripts/vad_bench/trim_regression/results/boundary_T0_T0.3.txt b/scripts/vad_bench/trim_regression/results/boundary_T0_T0.3.txt new file mode 100644 index 0000000..5c40c63 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/boundary_T0_T0.3.txt @@ -0,0 +1,36 @@ +== redux T0 -> T0.3 split=all kinds=['talk'] (errors by zone: A / B / B-A) + cutaway S: 0/ 0/ +0 D: 1/ 1/ +0 I: 1/ 0/ -1 total 2/ 1/ -1 + start-edge S: 9/ 7/ -2 D: 2/ 2/ +0 I: 7/ 9/ +2 total 18/ 18/ +0 + end-edge S: 4/ 5/ +1 D: 14/ 13/ -1 I: 3/ 3/ +0 total 21/ 21/ +0 + interior S: 545/ 555/ +10 D: 261/ 267/ +6 I: 186/ 188/ +2 total 992/1010/ +18 + {'tokens in cut-away': 1, ' of which error (S/I)': 1, 'tokens total (A)': 21459} +== redux T0 -> T0.3 split=all kinds=['sinr'] (errors by zone: A / B / B-A) + cutaway S: 0/ 0/ +0 D: 4/ 5/ +1 I: 0/ 0/ +0 total 4/ 5/ +1 + start-edge S: 16/ 6/ -10 D: 2/ 5/ +3 I: 3/ 4/ +1 total 21/ 15/ -6 + end-edge S: 30/ 12/ -18 D: 1/ 1/ +0 I: 0/ 0/ +0 total 31/ 13/ -18 + interior S: 2710/2664/ -46 D: 280/ 267/ -13 I: 218/ 238/ +20 total 3208/3169/ -39 + {'tokens total (A)': 49784, 'tokens in cut-away': 14, ' of which correct(C)': 14} +== ultra T0 -> T0.3 split=all kinds=['talk'] (errors by zone: A / B / B-A) + cutaway S: 0/ 0/ +0 D: 1/ 1/ +0 I: 0/ 0/ +0 total 1/ 1/ +0 + start-edge S: 10/ 8/ -2 D: 1/ 2/ +1 I: 9/ 11/ +2 total 20/ 21/ +1 + end-edge S: 3/ 3/ +0 D: 6/ 8/ +2 I: 2/ 3/ +1 total 11/ 14/ +3 + interior S: 437/ 441/ +4 D: 230/ 228/ -2 I: 185/ 181/ -4 total 852/ 850/ -2 + {'tokens total (A)': 21498, 'tokens in cut-away': 1, ' of which correct(C)': 1} +== ultra T0 -> T0.3 split=all kinds=['sinr'] (errors by zone: A / B / B-A) + cutaway S: 1/ 0/ -1 D: 11/ 4/ -7 I: 1/ 0/ -1 total 13/ 4/ -9 + start-edge S: 1/ 0/ -1 D: 0/ 0/ +0 I: 0/ 0/ +0 total 1/ 0/ -1 + end-edge S: 17/ 8/ -9 D: 0/ 3/ +3 I: 0/ 0/ +0 total 17/ 11/ -6 + interior S: 830/ 797/ -33 D: 123/ 122/ -1 I: 62/ 66/ +4 total 1015/ 985/ -30 + {'tokens total (A)': 17869, 'tokens in cut-away': 16, ' of which correct(C)': 14, ' of which error (S/I)': 2} +== v3 T0 -> T0.3 split=all kinds=['talk'] (errors by zone: A / B / B-A) + cutaway S: 0/ 0/ +0 D: 1/ 1/ +0 I: 0/ 0/ +0 total 1/ 1/ +0 + start-edge S: 12/ 11/ -1 D: 2/ 2/ +0 I: 17/ 16/ -1 total 31/ 29/ -2 + end-edge S: 2/ 2/ +0 D: 3/ 3/ +0 I: 3/ 3/ +0 total 8/ 8/ +0 + interior S: 426/ 423/ -3 D: 231/ 238/ +7 I: 213/ 213/ +0 total 870/ 874/ +4 + {'tokens total (A)': 21536} +== v3 T0 -> T0.3 split=all kinds=['sinr'] (errors by zone: A / B / B-A) + cutaway S: 3/ 0/ -3 D: 21/ 5/ -16 I: 0/ 0/ +0 total 24/ 5/ -19 + start-edge S: 1/ 4/ +3 D: 4/ 1/ -3 I: 0/ 0/ +0 total 5/ 5/ +0 + end-edge S: 16/ 10/ -6 D: 18/ 7/ -11 I: 0/ 0/ +0 total 34/ 17/ -17 + interior S: 835/ 821/ -14 D: 211/ 168/ -43 I: 66/ 67/ +1 total 1112/1056/ -56 + {'tokens total (A)': 17192, 'tokens in cut-away': 23, ' of which correct(C)': 20, ' of which error (S/I)': 3} diff --git a/scripts/vad_bench/trim_regression/results/cand_redux_heldout.md b/scripts/vad_bench/trim_regression/results/cand_redux_heldout.md new file mode 100644 index 0000000..50c7134 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/cand_redux_heldout.md @@ -0,0 +1,17 @@ +### redux, heldout: WER % (delta vs T0; paired bootstrap 95% CI; * = CI excludes 0) +| variant | talks | clean | noisy | pink0 | white0 | combined | +|---|---|---|---|---|---|---| +| T0 | 4.19 | 2.96 | 11.30 | 11.26 | 18.96 | 8.10 | +| T0.1 | 4.20 (+0.01; -0.20..+0.22) | 2.96 (+0.00; -0.30..+0.30) | 11.41 (+0.11; -0.36..+0.63) | 11.41 (+0.15; -0.91..+1.31) | 18.86 (-0.10; -1.11..+1.06) | 8.17 (+0.07; -0.22..+0.36) | +| T0.2 | 4.28 (+0.09; -0.17..+0.36) | 2.74 (-0.22; -0.67..+0.18) | 11.28 (-0.02; -0.45..+0.45) | 11.33 (+0.07; -0.89..+1.04) | 19.33 (+0.37; -0.64..+1.43) | 8.12 (+0.01; -0.25..+0.29) | +| T0.3 | 4.35 (+0.16; -0.01..+0.36) | 2.59 (-0.37; -0.80..-0.07)* | 11.28 (-0.02; -0.55..+0.64) | 11.36 (+0.10; -0.89..+1.25) | 18.84 (-0.12; -1.16..+0.97) | 8.14 (+0.03; -0.27..+0.41) | +| T0.32 | 4.27 (+0.08; -0.09..+0.29) | 2.81 (-0.15; -0.57..+0.15) | 11.29 (-0.01; -0.21..+0.19) | 11.41 (+0.15; -0.21..+0.51) | 19.26 (+0.30; -0.13..+0.74) | 8.12 (+0.02; -0.11..+0.16) | +| T0.35 | 4.21 (+0.02; -0.14..+0.17) | 2.81 (-0.15; -0.39..+0.00) | 11.31 (+0.01; -0.48..+0.51) | 11.36 (+0.10; -0.84..+1.06) | 19.06 (+0.10; -0.91..+1.10) | 8.11 (+0.01; -0.27..+0.28) | +| T0.5 | 4.14 (-0.05; -0.14..+0.03) | 2.96 (+0.00; -0.30..+0.31) | 11.41 (+0.10; -0.30..+0.53) | 11.38 (+0.12; -0.66..+0.91) | 19.36 (+0.40; -0.46..+1.35) | 8.14 (+0.04; -0.20..+0.28) | +| T1.0 | 4.19 (+0.00; +0.00..+0.00) | 2.96 (+0.00; -0.22..+0.21) | 11.25 (-0.06; -0.29..+0.18) | 11.41 (+0.15; -0.38..+0.70) | 18.89 (-0.07; -0.63..+0.47) | 8.07 (-0.03; -0.17..+0.10) | +| P0.5/0.3 | 4.15 (-0.04; -0.13..+0.04) | 2.96 (+0.00; -0.30..+0.30) | 11.33 (+0.03; -0.37..+0.47) | 11.41 (+0.15; -0.68..+1.01) | 19.01 (+0.05; -0.85..+1.07) | 8.10 (+0.00; -0.24..+0.25) | +| E0.5 | 4.19 (+0.00; +0.00..+0.00) | 2.89 (-0.07; -0.33..+0.16) | 11.28 (-0.02; -0.47..+0.58) | 11.26 (+0.00; -0.81..+1.03) | 19.04 (+0.07; -0.49..+0.66) | 8.09 (-0.02; -0.26..+0.32) | +| E1.0 | 4.19 (+0.00; +0.00..+0.00) | 2.89 (-0.07; -0.23..+0.00) | 11.13 (-0.17; -0.36..-0.01)* | 10.99 (-0.27; -0.66..+0.08) | 18.96 (+0.00; -0.39..+0.34) | 8.00 (-0.10; -0.20..-0.01)* | +| E0.5p0.5 | 4.19 (+0.00; +0.00..+0.00) | 2.96 (+0.00; -0.21..+0.21) | 11.31 (+0.01; -0.22..+0.24) | 11.38 (+0.12; -0.27..+0.55) | 18.91 (-0.05; -0.48..+0.35) | 8.11 (+0.00; -0.13..+0.13) | +| L3 | 4.33 (+0.14; -0.02..+0.33) | 2.59 (-0.37; -0.80..-0.07)* | 11.28 (-0.02; -0.55..+0.64) | 11.33 (+0.07; -0.91..+1.21) | 18.84 (-0.12; -1.16..+0.97) | 8.13 (+0.03; -0.28..+0.40) | +| ref words | 11490 | 1350 | 16200 | 4050 | 4050 | 29040 | diff --git a/scripts/vad_bench/trim_regression/results/cand_redux_tune.md b/scripts/vad_bench/trim_regression/results/cand_redux_tune.md new file mode 100644 index 0000000..63c379d --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/cand_redux_tune.md @@ -0,0 +1,17 @@ +### redux, tune: WER % (delta vs T0; paired bootstrap 95% CI; * = CI excludes 0) +| variant | talks | clean | noisy | pink0 | white0 | combined | +|---|---|---|---|---|---|---| +| T0 | 5.48 | 1.40 | 8.14 | 8.90 | 14.86 | 6.45 | +| T0.1 | 5.53 (+0.05; -0.16..+0.26) | 1.55 (+0.16; -0.37..+0.76) | 7.43 (-0.71; -1.34..-0.12)* | 7.71 (-1.19; -2.24..-0.24)* | 13.87 (-0.98; -2.42..+0.43) | 6.19 (-0.27; -0.56..+0.03) | +| T0.2 | 5.43 (-0.05; -0.22..+0.12) | 1.40 (+0.00; -0.47..+0.46) | 8.02 (-0.12; -0.87..+0.77) | 7.25 (-1.66; -2.78..-0.56)* | 16.25 (+1.40; -0.92..+4.11) | 6.38 (-0.08; -0.41..+0.31) | +| T0.3 | 5.47 (-0.01; -0.14..+0.13) | 1.24 (-0.16; -0.75..+0.35) | 7.82 (-0.32; -0.85..+0.26) | 8.23 (-0.67; -1.78..+0.29) | 14.75 (-0.10; -1.51..+1.41) | 6.31 (-0.15; -0.38..+0.11) | +| T0.32 | 5.45 (-0.03; -0.13..+0.08) | 1.40 (+0.00; -0.49..+0.47) | 8.04 (-0.10; -0.32..+0.10) | 8.44 (-0.47; -1.04..-0.05)* | 14.96 (+0.10; -0.59..+0.91) | 6.39 (-0.06; -0.17..+0.05) | +| T0.35 | 5.48 (+0.00; -0.12..+0.13) | 1.55 (+0.16; -0.47..+0.97) | 7.82 (-0.32; -0.92..+0.26) | 7.76 (-1.14; -2.43..-0.08)* | 15.22 (+0.36; -1.26..+2.17) | 6.32 (-0.13; -0.37..+0.12) | +| T0.5 | 5.49 (+0.01; -0.07..+0.12) | 1.24 (-0.16; -0.56..+0.00) | 8.13 (-0.01; -0.33..+0.34) | 8.23 (-0.67; -1.34..-0.07)* | 15.73 (+0.88; -0.09..+2.15) | 6.45 (-0.01; -0.15..+0.15) | +| T1.0 | 5.47 (-0.01; -0.03..+0.00) | 1.55 (+0.16; +0.00..+0.56) | 8.31 (+0.17; -0.21..+0.70) | 8.70 (-0.21; -0.86..+0.47) | 15.68 (+0.83; -0.22..+2.24) | 6.52 (+0.07; -0.08..+0.31) | +| P0.5/0.3 | 5.45 (-0.03; -0.13..+0.08) | 1.40 (+0.00; -0.47..+0.46) | 8.00 (-0.14; -0.45..+0.13) | 8.13 (-0.78; -1.50..-0.16)* | 15.53 (+0.67; -0.08..+1.52) | 6.38 (-0.08; -0.22..+0.06) | +| E0.5 | 5.48 (+0.00; -0.10..+0.11) | 1.09 (-0.31; -0.86..+0.00) | 8.36 (+0.22; -0.14..+0.68) | 9.06 (+0.16; -0.46..+0.77) | 15.89 (+1.04; +0.05..+2.37)* | 6.54 (+0.08; -0.08..+0.28) | +| E1.0 | 5.48 (+0.00; +0.00..+0.00) | 1.24 (-0.16; -0.56..+0.00) | 8.15 (+0.01; -0.11..+0.14) | 8.80 (-0.10; -0.55..+0.31) | 15.06 (+0.21; -0.12..+0.64) | 6.45 (+0.00; -0.06..+0.06) | +| E0.5p0.5 | 5.47 (-0.01; -0.03..+0.00) | 1.24 (-0.16; -0.56..+0.00) | 8.14 (+0.00; -0.19..+0.26) | 8.54 (-0.36; -0.76..-0.09)* | 15.37 (+0.52; -0.14..+1.61) | 6.44 (-0.01; -0.10..+0.09) | +| L3 | 5.47 (-0.01; -0.14..+0.13) | 1.24 (-0.16; -0.75..+0.35) | 7.82 (-0.32; -0.85..+0.26) | 8.23 (-0.67; -1.78..+0.29) | 14.75 (-0.10; -1.51..+1.41) | 6.31 (-0.15; -0.38..+0.11) | +| ref words | 10050 | 644 | 7728 | 1932 | 1932 | 18422 | diff --git a/scripts/vad_bench/trim_regression/results/cand_ultra_heldout.md b/scripts/vad_bench/trim_regression/results/cand_ultra_heldout.md new file mode 100644 index 0000000..4f5873b --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/cand_ultra_heldout.md @@ -0,0 +1,9 @@ +### ultra, heldout: WER % (delta vs T0; paired bootstrap 95% CI; * = CI excludes 0) +| variant | talks | clean | noisy | pink0 | white0 | combined | +|---|---|---|---|---|---|---| +| T0 | 3.59 | 3.11 | 6.76 | 8.77 | - | 5.10 | +| T0.3 | 3.60 (+0.02; -0.06..+0.11) | 2.89 (-0.22; -0.76..+0.28) | 6.58 (-0.17; -0.49..+0.11) | 8.62 (-0.15; -0.64..+0.36) | - | 5.01 (-0.09; -0.27..+0.07) | +| T0.5 | 3.55 (-0.03; -0.09..+0.01) | 2.89 (-0.22; -0.71..+0.24) | 6.72 (-0.04; -0.33..+0.26) | 8.74 (-0.02; -0.57..+0.57) | - | 5.05 (-0.05; -0.21..+0.10) | +| P0.5/0.3 | 3.55 (-0.03; -0.09..+0.01) | 2.89 (-0.22; -0.71..+0.24) | 6.72 (-0.04; -0.35..+0.27) | 8.72 (-0.05; -0.59..+0.57) | - | 5.05 (-0.05; -0.23..+0.11) | +| E0.5 | 3.59 (+0.00; +0.00..+0.00) | 2.74 (-0.37; -0.87..+0.07) | 6.67 (-0.09; -0.34..+0.13) | 8.69 (-0.07; -0.37..+0.21) | - | 5.04 (-0.06; -0.21..+0.06) | +| ref words | 11490 | 1350 | 12150 | 4050 | | 24990 | diff --git a/scripts/vad_bench/trim_regression/results/cand_ultra_tune.md b/scripts/vad_bench/trim_regression/results/cand_ultra_tune.md new file mode 100644 index 0000000..2183015 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/cand_ultra_tune.md @@ -0,0 +1,9 @@ +### ultra, tune: WER % (delta vs T0; paired bootstrap 95% CI; * = CI excludes 0) +| variant | talks | clean | noisy | pink0 | white0 | combined | +|---|---|---|---|---|---|---| +| T0 | 4.70 | 2.06 | 4.67 | 7.65 | - | 4.62 | +| T0.3 | 4.70 (+0.00; -0.10..+0.12) | 2.06 (+0.00; +0.00..+0.00) | 4.04 (-0.63; -1.37..+0.03) | 6.53 (-1.12; -2.53..+0.18) | - | 4.46 (-0.16; -0.37..+0.04) | +| T0.5 | 4.75 (+0.05; +0.00..+0.12) | 0.77 (-1.29; -2.90..+0.00) | 4.35 (-0.32; -0.93..+0.28) | 6.70 (-0.95; -2.33..+0.30) | - | 4.54 (-0.08; -0.26..+0.10) | +| P0.5/0.3 | 4.74 (+0.04; -0.01..+0.11) | 0.77 (-1.29; -2.90..+0.00) | 4.35 (-0.32; -0.88..+0.23) | 6.79 (-0.86; -2.23..+0.32) | - | 4.53 (-0.09; -0.26..+0.08) | +| E0.5 | 4.70 (+0.00; -0.03..+0.03) | 2.06 (+0.00; +0.00..+0.00) | 4.55 (-0.11; -0.55..+0.28) | 7.47 (-0.17; -1.31..+0.74) | - | 4.59 (-0.03; -0.15..+0.07) | +| ref words | 10050 | 388 | 3492 | 1164 | | 13930 | diff --git a/scripts/vad_bench/trim_regression/results/cand_v3_heldout.md b/scripts/vad_bench/trim_regression/results/cand_v3_heldout.md new file mode 100644 index 0000000..c19781f --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/cand_v3_heldout.md @@ -0,0 +1,9 @@ +### v3, heldout: WER % (delta vs T0; paired bootstrap 95% CI; * = CI excludes 0) +| variant | talks | clean | noisy | pink0 | white0 | combined | +|---|---|---|---|---|---|---| +| T0 | 3.59 | 6.37 | 7.62 | 10.69 | - | 5.70 | +| T0.3 | 3.62 (+0.03; -0.08..+0.20) | 3.19 (-3.19; -7.76..-0.14)* | 7.33 (-0.29; -0.61..+0.03) | 10.54 (-0.15; -0.78..+0.51) | - | 5.40 (-0.30; -0.63..-0.03)* | +| T0.5 | 3.57 (-0.02; -0.04..+0.00) | 3.11 (-3.26; -7.91..-0.18)* | 7.63 (+0.01; -0.31..+0.33) | 10.67 (-0.02; -0.75..+0.65) | - | 5.52 (-0.18; -0.47..+0.05) | +| P0.5/0.3 | 3.59 (+0.01; -0.04..+0.06) | 3.11 (-3.26; -7.91..-0.18)* | 7.58 (-0.04; -0.36..+0.27) | 10.62 (-0.07; -0.88..+0.65) | - | 5.51 (-0.19; -0.49..+0.05) | +| E0.5 | 3.59 (+0.00; +0.00..+0.00) | 3.11 (-3.26; -7.85..-0.21)* | 7.38 (-0.24; -0.59..+0.08) | 10.54 (-0.15; -0.64..+0.42) | - | 5.41 (-0.29; -0.66..-0.02)* | +| ref words | 11490 | 1350 | 12150 | 4050 | | 24990 | diff --git a/scripts/vad_bench/trim_regression/results/cand_v3_tune.md b/scripts/vad_bench/trim_regression/results/cand_v3_tune.md new file mode 100644 index 0000000..963ef70 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/cand_v3_tune.md @@ -0,0 +1,9 @@ +### v3, tune: WER % (delta vs T0; paired bootstrap 95% CI; * = CI excludes 0) +| variant | talks | clean | noisy | pink0 | white0 | combined | +|---|---|---|---|---|---|---| +| T0 | 4.96 | 3.09 | 4.32 | 6.96 | - | 4.75 | +| T0.3 | 4.94 (-0.02; -0.09..+0.04) | 1.03 (-2.06; -7.62..+0.00) | 4.15 (-0.17; -0.83..+0.35) | 6.70 (-0.26; -1.27..+0.92) | - | 4.63 (-0.11; -0.32..+0.05) | +| T0.5 | 4.97 (+0.01; -0.04..+0.07) | 1.03 (-2.06; -7.62..+0.00) | 4.50 (+0.17; -0.13..+0.53) | 7.39 (+0.43; -0.47..+1.65) | - | 4.74 (-0.01; -0.17..+0.12) | +| P0.5/0.3 | 4.98 (+0.02; -0.03..+0.08) | 1.03 (-2.06; -7.62..+0.00) | 4.52 (+0.20; -0.06..+0.54) | 7.65 (+0.69; -0.16..+1.88) | - | 4.75 (+0.01; -0.15..+0.13) | +| E0.5 | 4.97 (+0.01; +0.00..+0.03) | 1.03 (-2.06; -7.62..+0.00) | 4.27 (-0.06; -0.61..+0.45) | 6.62 (-0.34; -1.40..+0.90) | - | 4.68 (-0.06; -0.25..+0.09) | +| ref words | 10050 | 388 | 3492 | 1164 | | 13930 | diff --git a/scripts/vad_bench/trim_regression/results/clip_count.txt b/scripts/vad_bench/trim_regression/results/clip_count.txt new file mode 100644 index 0000000..bb76e35 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/clip_count.txt @@ -0,0 +1,66 @@ +== ultra + talks T0 words 21498 + T0.1 cut correct 11 cut error 1 | touched correct 112 touched error 5 + T0.2 cut correct 2 cut error 1 | touched correct 48 touched error 1 + T0.3 cut correct 1 cut error 0 | touched correct 17 touched error 1 + T0.5 cut correct 0 cut error 0 | touched correct 6 touched error 0 + T1.0 cut correct 0 cut error 0 | touched correct 5 touched error 0 + P0.5/0.3 cut correct 1 cut error 0 | touched correct 12 touched error 0 + sinr noisy T0 words 16133 + T0.1 cut correct 103 cut error 3 | touched correct 298 touched error 9 + T0.2 cut correct 31 cut error 3 | touched correct 239 touched error 6 + T0.3 cut correct 14 cut error 2 | touched correct 161 touched error 5 + T0.5 cut correct 1 cut error 2 | touched correct 24 touched error 2 + T1.0 cut correct 0 cut error 1 | touched correct 0 touched error 2 + P0.5/0.3 cut correct 4 cut error 2 | touched correct 73 touched error 4 + sinr clean T0 words 1736 + T0.1 cut correct 5 cut error 0 | touched correct 27 touched error 1 + T0.2 cut correct 1 cut error 0 | touched correct 18 touched error 1 + T0.3 cut correct 0 cut error 0 | touched correct 6 touched error 0 + T0.5 cut correct 0 cut error 0 | touched correct 0 touched error 0 + T1.0 cut correct 0 cut error 0 | touched correct 0 touched error 0 + P0.5/0.3 cut correct 0 cut error 0 | touched correct 1 touched error 0 +== redux + talks T0 words 21459 + T0.1 cut correct 16 cut error 2 | touched correct 107 touched error 6 + T0.2 cut correct 1 cut error 1 | touched correct 37 touched error 2 + T0.3 cut correct 0 cut error 1 | touched correct 14 touched error 1 + T0.5 cut correct 0 cut error 0 | touched correct 10 touched error 0 + T1.0 cut correct 0 cut error 0 | touched correct 10 touched error 0 + P0.5/0.3 cut correct 0 cut error 0 | touched correct 11 touched error 0 + sinr noisy T0 words 47791 + T0.1 cut correct 226 cut error 2 | touched correct 442 touched error 40 + T0.2 cut correct 63 cut error 1 | touched correct 304 touched error 25 + T0.3 cut correct 14 cut error 0 | touched correct 109 touched error 7 + T0.5 cut correct 0 cut error 0 | touched correct 10 touched error 0 + T1.0 cut correct 0 cut error 0 | touched correct 0 touched error 0 + P0.5/0.3 cut correct 0 cut error 0 | touched correct 19 touched error 6 + sinr clean T0 words 1993 + T0.1 cut correct 10 cut error 0 | touched correct 19 touched error 1 + T0.2 cut correct 2 cut error 0 | touched correct 13 touched error 1 + T0.3 cut correct 0 cut error 0 | touched correct 2 touched error 0 + T0.5 cut correct 0 cut error 0 | touched correct 0 touched error 0 + T1.0 cut correct 0 cut error 0 | touched correct 0 touched error 0 + P0.5/0.3 cut correct 0 cut error 0 | touched correct 0 touched error 0 +== v3 + talks T0 words 21536 + T0.1 cut correct 20 cut error 3 | touched correct 95 touched error 10 + T0.2 cut correct 4 cut error 0 | touched correct 37 touched error 4 + T0.3 cut correct 0 cut error 0 | touched correct 10 touched error 0 + T0.5 cut correct 0 cut error 0 | touched correct 4 touched error 0 + T1.0 cut correct 0 cut error 0 | touched correct 4 touched error 0 + P0.5/0.3 cut correct 0 cut error 0 | touched correct 6 touched error 0 + sinr noisy T0 words 15506 + T0.1 cut correct 152 cut error 4 | touched correct 324 touched error 15 + T0.2 cut correct 61 cut error 3 | touched correct 245 touched error 6 + T0.3 cut correct 19 cut error 3 | touched correct 175 touched error 5 + T0.5 cut correct 4 cut error 2 | touched correct 17 touched error 2 + T1.0 cut correct 1 cut error 0 | touched correct 2 touched error 0 + P0.5/0.3 cut correct 7 cut error 3 | touched correct 58 touched error 5 + sinr clean T0 words 1686 + T0.1 cut correct 9 cut error 0 | touched correct 32 touched error 0 + T0.2 cut correct 2 cut error 0 | touched correct 21 touched error 0 + T0.3 cut correct 1 cut error 0 | touched correct 6 touched error 0 + T0.5 cut correct 0 cut error 0 | touched correct 0 touched error 0 + T1.0 cut correct 0 cut error 0 | touched correct 0 touched error 0 + P0.5/0.3 cut correct 0 cut error 0 | touched correct 1 touched error 0 diff --git a/scripts/vad_bench/trim_regression/results/grid_redux_all.md b/scripts/vad_bench/trim_regression/results/grid_redux_all.md new file mode 100644 index 0000000..e094976 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/grid_redux_all.md @@ -0,0 +1,10 @@ +### redux, all: WER % (delta vs T0, paired bootstrap 95% CI) +| variant | talks | clean | white | pink | pink0 | white0 | noisy-all | sinr-all | +|---|---|---|---|---|---|---|---|---| +| T0 | 4.80 | 2.46 | 8.23 | 5.20 | 10.50 | 17.64 | 6.72 | 6.55 | +| T0.1 | 4.82 (+0.03; -0.12..+0.18) | 2.51 (+0.05; -0.21..+0.32) | 8.11 (-0.12; -0.38..+0.16) | 5.31 (+0.11; -0.16..+0.40) | 10.21 (-0.28; -1.07..+0.56) | 17.25 (-0.38; -1.19..+0.51) | 6.71 (-0.01; -0.22..+0.22) | 6.54 (-0.00; -0.21..+0.21) | +| T0.2 | 4.82 (+0.02; -0.13..+0.19) | 2.31 (-0.15; -0.48..+0.16) | 8.19 (-0.05; -0.41..+0.33) | 5.03 (-0.18; -0.47..+0.12) | 10.01 (-0.48; -1.24..+0.28) | 18.34 (+0.70; -0.30..+1.83) | 6.61 (-0.11; -0.40..+0.16) | 6.44 (-0.11; -0.39..+0.16) | +| T0.3 | 4.87 (+0.08; -0.03..+0.20) | 2.16 (-0.30; -0.65..-0.04)* | 8.07 (-0.16; -0.52..+0.18) | 5.13 (-0.08; -0.40..+0.27) | 10.35 (-0.15; -0.90..+0.71) | 17.52 (-0.12; -0.93..+0.76) | 6.60 (-0.12; -0.42..+0.17) | 6.42 (-0.12; -0.42..+0.16) | +| T0.5 | 4.77 (-0.02; -0.08..+0.04) | 2.41 (-0.05; -0.27..+0.17) | 8.32 (+0.09; -0.13..+0.32) | 5.24 (+0.03; -0.21..+0.29) | 10.36 (-0.13; -0.71..+0.43) | 18.19 (+0.55; -0.11..+1.30) | 6.78 (+0.06; -0.13..+0.25) | 6.60 (+0.06; -0.13..+0.25) | +| T1.0 | 4.79 (-0.00; -0.01..+0.00) | 2.51 (+0.05; -0.11..+0.23) | 8.20 (-0.03; -0.19..+0.15) | 5.21 (+0.00; -0.14..+0.14) | 10.53 (+0.03; -0.39..+0.46) | 17.85 (+0.22; -0.31..+0.82) | 6.71 (-0.01; -0.13..+0.12) | 6.54 (-0.01; -0.13..+0.11) | +| words in reference | 21540 | 1994 | 23928 | 23928 | 5982 | 5982 | 47856 | 49850 | diff --git a/scripts/vad_bench/trim_regression/results/grid_redux_heldout.md b/scripts/vad_bench/trim_regression/results/grid_redux_heldout.md new file mode 100644 index 0000000..beb0cc2 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/grid_redux_heldout.md @@ -0,0 +1,10 @@ +### redux, heldout: WER % (delta vs T0, paired bootstrap 95% CI) +| variant | talks | clean | white | pink | pink0 | white0 | noisy-all | sinr-all | +|---|---|---|---|---|---|---|---|---| +| T0 | 4.19 | 2.96 | 9.18 | 5.77 | 11.26 | 18.96 | 7.48 | 7.29 | +| T0.1 | 4.20 (+0.01; -0.20..+0.22) | 2.96 (+0.00; -0.30..+0.30) | 9.19 (+0.01; -0.28..+0.36) | 6.05 (+0.28; -0.08..+0.66) | 11.41 (+0.15; -0.91..+1.31) | 18.86 (-0.10; -1.11..+1.06) | 7.62 (+0.14; -0.10..+0.41) | 7.43 (+0.14; -0.10..+0.39) | +| T0.2 | 4.28 (+0.09; -0.17..+0.36) | 2.74 (-0.22; -0.67..+0.18) | 8.94 (-0.23; -0.69..+0.18) | 5.69 (-0.08; -0.45..+0.29) | 11.33 (+0.07; -0.89..+1.04) | 19.33 (+0.37; -0.64..+1.43) | 7.32 (-0.16; -0.54..+0.20) | 7.13 (-0.16; -0.53..+0.18) | +| T0.3 | 4.35 (+0.16; -0.01..+0.36) | 2.59 (-0.37; -0.80..-0.07)* | 9.03 (-0.15; -0.68..+0.32) | 5.75 (-0.02; -0.44..+0.45) | 11.36 (+0.10; -0.89..+1.25) | 18.84 (-0.12; -1.16..+0.97) | 7.39 (-0.08; -0.50..+0.31) | 7.20 (-0.09; -0.51..+0.29) | +| T0.5 | 4.14 (-0.05; -0.14..+0.03) | 2.96 (+0.00; -0.30..+0.31) | 9.19 (+0.01; -0.26..+0.31) | 5.93 (+0.16; -0.16..+0.51) | 11.38 (+0.12; -0.66..+0.91) | 19.36 (+0.40; -0.46..+1.35) | 7.56 (+0.09; -0.16..+0.35) | 7.38 (+0.08; -0.16..+0.34) | +| T1.0 | 4.19 (+0.00; +0.00..+0.00) | 2.96 (+0.00; -0.22..+0.21) | 9.02 (-0.15; -0.32..+0.01) | 5.84 (+0.07; -0.12..+0.27) | 11.41 (+0.15; -0.38..+0.70) | 18.89 (-0.07; -0.63..+0.47) | 7.43 (-0.04; -0.18..+0.09) | 7.25 (-0.04; -0.17..+0.08) | +| words in reference | 11490 | 1350 | 16200 | 16200 | 4050 | 4050 | 32400 | 33750 | diff --git a/scripts/vad_bench/trim_regression/results/grid_redux_tune.md b/scripts/vad_bench/trim_regression/results/grid_redux_tune.md new file mode 100644 index 0000000..153cf37 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/grid_redux_tune.md @@ -0,0 +1,10 @@ +### redux, tune: WER % (delta vs T0, paired bootstrap 95% CI) +| variant | talks | clean | white | pink | pink0 | white0 | noisy-all | sinr-all | +|---|---|---|---|---|---|---|---|---| +| T0 | 5.48 | 1.40 | 6.25 | 4.01 | 8.90 | 14.86 | 5.13 | 4.98 | +| T0.1 | 5.53 (+0.05; -0.16..+0.26) | 1.55 (+0.16; -0.37..+0.76) | 5.86 (-0.39; -0.91..+0.10) | 3.77 (-0.25; -0.61..+0.14) | 7.71 (-1.19; -2.24..-0.24)* | 13.87 (-0.98; -2.42..+0.43) | 4.81 (-0.32; -0.67..+0.04) | 4.68 (-0.30; -0.65..+0.05) | +| T0.2 | 5.43 (-0.05; -0.22..+0.12) | 1.40 (+0.00; -0.47..+0.46) | 6.60 (+0.35; -0.26..+1.09) | 3.64 (-0.38; -0.81..+0.05) | 7.25 (-1.66; -2.78..-0.56)* | 16.25 (+1.40; -0.92..+4.11) | 5.12 (-0.01; -0.43..+0.47) | 4.97 (-0.01; -0.42..+0.45) | +| T0.3 | 5.47 (-0.01; -0.14..+0.13) | 1.24 (-0.16; -0.75..+0.35) | 6.07 (-0.18; -0.60..+0.27) | 3.82 (-0.19; -0.66..+0.25) | 8.23 (-0.67; -1.78..+0.29) | 14.75 (-0.10; -1.51..+1.41) | 4.94 (-0.19; -0.55..+0.19) | 4.80 (-0.19; -0.55..+0.19) | +| T0.5 | 5.49 (+0.01; -0.07..+0.12) | 1.24 (-0.16; -0.56..+0.00) | 6.50 (+0.25; -0.08..+0.61) | 3.78 (-0.23; -0.52..+0.04) | 8.23 (-0.67; -1.34..-0.07)* | 15.73 (+0.88; -0.09..+2.15) | 5.14 (+0.01; -0.24..+0.26) | 4.98 (+0.00; -0.24..+0.25) | +| T1.0 | 5.47 (-0.01; -0.03..+0.00) | 1.55 (+0.16; +0.00..+0.56) | 6.48 (+0.23; -0.07..+0.63) | 3.88 (-0.13; -0.34..+0.08) | 8.70 (-0.21; -0.86..+0.47) | 15.68 (+0.83; -0.22..+2.24) | 5.18 (+0.05; -0.17..+0.33) | 5.04 (+0.06; -0.15..+0.32) | +| words in reference | 10050 | 644 | 7728 | 7728 | 1932 | 1932 | 15456 | 16100 | diff --git a/scripts/vad_bench/trim_regression/results/grid_ultra_talks_all.txt b/scripts/vad_bench/trim_regression/results/grid_ultra_talks_all.txt new file mode 100644 index 0000000..b38d7f9 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/grid_ultra_talks_all.txt @@ -0,0 +1 @@ +talks[all] n_ref= 21540 units= 220 | 4.10 [450/238/196] | 4.19 +0.08(-0.04,+0.22) [460/237/205] | 4.07 -0.03(-0.13,+0.06) [451/232/194] | 4.11 +0.01(-0.06,+0.08) [452/239/195] | 4.11 +0.00(-0.03,+0.05) [452/237/196] | 4.10 -0.00(-0.01,+0.00) [450/238/195] diff --git a/scripts/vad_bench/trim_regression/results/grid_v3_talks_all.txt b/scripts/vad_bench/trim_regression/results/grid_v3_talks_all.txt new file mode 100644 index 0000000..fbe5794 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/grid_v3_talks_all.txt @@ -0,0 +1 @@ +talks[all] n_ref= 21540 units= 220 | 4.22 [440/237/233] | 4.30 +0.08(-0.07,+0.23) [455/237/235] | 4.18 -0.05(-0.14,+0.03) [436/228/236] | 4.23 +0.01(-0.06,+0.10) [436/244/232] | 4.22 -0.00(-0.03,+0.02) [441/235/233] | 4.22 +0.00(+0.00,+0.00) [440/237/233] diff --git a/scripts/vad_bench/trim_regression/results/noise_seconds_90_inserts.txt b/scripts/vad_bench/trim_regression/results/noise_seconds_90_inserts.txt new file mode 100644 index 0000000..d2f0990 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/noise_seconds_90_inserts.txt @@ -0,0 +1,35 @@ +90 insert files (9 talks x white,pink,clicks,music @-20; white,music @-35; white,pink,clicks,music @-5) +det variant sec/file +ultra T0 37.2 clicks-20:28 clicks-5:56 music-20:21 music-35:27 music-5:23 pink-20:27 pink-5:60 white-20:45 white-35:26 white-5:60 +ultra T0.1 11.7 clicks-20:0 clicks-5:7 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:51 white-20:1 white-35:0 white-5:59 +ultra T0.3 12.1 clicks-20:0 clicks-5:8 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:51 white-20:1 white-35:0 white-5:59 +ultra T0.5 12.4 clicks-20:0 clicks-5:9 music-20:0 music-35:0 music-5:1 pink-20:0 pink-5:52 white-20:1 white-35:0 white-5:60 +ultra T1.0 13.3 clicks-20:1 clicks-5:11 music-20:1 music-35:1 music-5:1 pink-20:1 pink-5:54 white-20:3 white-35:1 white-5:60 +ultra P0.5/0.3 12.2 clicks-20:0 clicks-5:8 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:52 white-20:1 white-35:0 white-5:60 +ultra E0.5 12.2 clicks-20:0 clicks-5:8 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:52 white-20:1 white-35:0 white-5:60 +ultra E1.0 12.3 clicks-20:0 clicks-5:8 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:53 white-20:1 white-35:0 white-5:60 +ultra E0.5p0.5 12.5 clicks-20:0 clicks-5:9 music-20:0 music-35:0 music-5:1 pink-20:0 pink-5:52 white-20:1 white-35:0 white-5:60 +ultra L3 13.0 clicks-20:1 clicks-5:11 music-20:1 music-35:1 music-5:1 pink-20:1 pink-5:51 white-20:3 white-35:1 white-5:59 +ultra L6 14.4 clicks-20:2 clicks-5:16 music-20:1 music-35:2 music-5:2 pink-20:2 pink-5:52 white-20:7 white-35:2 white-5:59 +redux T0 33.2 clicks-20:26 clicks-5:26 music-20:28 music-35:37 music-5:22 pink-20:33 pink-5:60 white-20:29 white-35:26 white-5:46 +redux T0.1 4.4 clicks-20:0 clicks-5:0 music-20:7 music-35:9 music-5:0 pink-20:0 pink-5:17 white-20:0 white-35:0 white-5:11 +redux T0.3 4.6 clicks-20:0 clicks-5:0 music-20:7 music-35:9 music-5:0 pink-20:0 pink-5:18 white-20:0 white-35:0 white-5:12 +redux T0.5 4.9 clicks-20:0 clicks-5:0 music-20:7 music-35:9 music-5:0 pink-20:0 pink-5:18 white-20:0 white-35:0 white-5:12 +redux T1.0 5.7 clicks-20:1 clicks-5:1 music-20:8 music-35:10 music-5:1 pink-20:1 pink-5:20 white-20:1 white-35:1 white-5:13 +redux P0.5/0.3 4.7 clicks-20:0 clicks-5:0 music-20:7 music-35:9 music-5:0 pink-20:0 pink-5:18 white-20:0 white-35:0 white-5:12 +redux E0.5 4.6 clicks-20:0 clicks-5:0 music-20:7 music-35:9 music-5:0 pink-20:0 pink-5:18 white-20:0 white-35:0 white-5:12 +redux E1.0 4.7 clicks-20:0 clicks-5:0 music-20:7 music-35:9 music-5:0 pink-20:0 pink-5:18 white-20:0 white-35:0 white-5:12 +redux E0.5p0.5 4.9 clicks-20:0 clicks-5:0 music-20:7 music-35:9 music-5:0 pink-20:0 pink-5:18 white-20:0 white-35:0 white-5:12 +redux L3 5.3 clicks-20:0 clicks-5:0 music-20:8 music-35:9 music-5:0 pink-20:1 pink-5:20 white-20:1 white-35:1 white-5:13 +redux L6 6.7 clicks-20:1 clicks-5:1 music-20:9 music-35:11 music-5:1 pink-20:3 pink-5:23 white-20:2 white-35:1 white-5:15 +v3 T0 27.0 clicks-20:27 clicks-5:27 music-20:27 music-35:27 music-5:27 pink-20:27 pink-5:27 white-20:27 white-35:27 white-5:27 +v3 T0.1 0.0 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 T0.3 0.0 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 T0.5 0.2 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 T1.0 0.6 clicks-20:1 clicks-5:1 music-20:1 music-35:1 music-5:1 pink-20:1 pink-5:1 white-20:1 white-35:1 white-5:1 +v3 P0.5/0.3 0.0 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 E0.5 0.1 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 E1.0 0.1 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 E0.5p0.5 0.2 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 L3 0.5 clicks-20:0 clicks-5:0 music-20:0 music-35:0 music-5:0 pink-20:0 pink-5:0 white-20:0 white-35:0 white-5:0 +v3 L6 1.6 clicks-20:2 clicks-5:2 music-20:2 music-35:2 music-5:2 pink-20:2 pink-5:2 white-20:2 white-35:2 white-5:2 diff --git a/scripts/vad_bench/trim_regression/results/noise_words_20_inserts.md b/scripts/vad_bench/trim_regression/results/noise_words_20_inserts.md new file mode 100644 index 0000000..a325781 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/noise_words_20_inserts.md @@ -0,0 +1,21 @@ +20 files, noise block 60 s each +| detector | variant | noise sec decoded / file | invented words | files with words | words per file with noise decoded | +|---|---|---|---|---|---| +| redux | T0 | 38.3 | 0 | 0/20 | | +| redux | T0.3 | 5.3 | 0 | 0/20 | | +| redux | T0.5 | 5.7 | 0 | 0/20 | | +| redux | P0.5/0.3 | 5.4 | 0 | 0/20 | | +| redux | E0.5 | 5.4 | 0 | 0/20 | | +| redux | L3 | 6.4 | 0 | 0/20 | | +| ultra | T0 | 50.9 | 18 | 2/20 | | +| ultra | T0.3 | 22.7 | 0 | 0/20 | | +| ultra | T0.5 | 23.3 | 9 | 1/20 | | +| ultra | P0.5/0.3 | 23.0 | 9 | 1/20 | | +| ultra | E0.5 | 22.9 | 9 | 1/20 | | +| ultra | L3 | 24.1 | 0 | 0/20 | | +| v3 | T0 | 26.8 | 14 | 2/20 | | +| v3 | T0.3 | 0.1 | 0 | 0/20 | | +| v3 | T0.5 | 0.2 | 0 | 0/20 | | +| v3 | P0.5/0.3 | 0.1 | 0 | 0/20 | | +| v3 | E0.5 | 0.1 | 0 | 0/20 | | +| v3 | L3 | 0.5 | 6 | 1/20 | | diff --git a/scripts/vad_bench/trim_regression/results/per_detector_default_table.md b/scripts/vad_bench/trim_regression/results/per_detector_default_table.md new file mode 100644 index 0000000..29bb774 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/per_detector_default_table.md @@ -0,0 +1,19 @@ +| detector | rule (pad before / after) | tune WER (talks+clean+3 noisy) | held-out WER | held-out delta vs T0.3 (95% CI) | noise s decoded of 60 s (90 files) | invented words in block (20 files) | +|---|---|---|---|---|---|---| +| redux | - (no trim) | 5.47 | 6.34 | -0.06 (-0.46..+0.26) | 33.2 | 0 | +| redux | 0.3 / 0.3 | 5.32 | 6.40 | - | 4.6 | 0 | +| redux | 0.5 / 0.5 | 5.36 | 6.32 | -0.08 (-0.43..+0.24) | 4.9 | 0 | +| redux | 0.5 / 0.3 | 5.31 | 6.33 | -0.07 (-0.42..+0.24) | 4.7 | 0 | +| redux | 0.3 / 0.3, only edges with >=0.5 s removed | 5.44 | 6.31 | -0.09 (-0.25..+0.05) | 4.6 | 0 | +| ultra | - (no trim) | 4.62 | 5.10 | +0.09 (-0.07..+0.27) | 37.2 | 18 | +| ultra | 0.3 / 0.3 | 4.46 | 5.01 | - | 12.1 | 0 | +| ultra | 0.5 / 0.5 | 4.54 | 5.05 | +0.04 (-0.10..+0.17) | 12.4 | 9 | +| ultra | 0.5 / 0.3 | 4.53 | 5.05 | +0.04 (-0.09..+0.16) | 12.2 | 9 | +| ultra | 0.3 / 0.3, only edges with >=0.5 s removed | 4.59 | 5.04 | +0.02 (-0.07..+0.12) | 12.2 | 9 | +| v3 | - (no trim) | 4.75 | 5.70 | +0.30 (+0.03..+0.63) | 27.0 | 14 | +| v3 | 0.3 / 0.3 | 4.63 | 5.40 | - | 0.0 | 0 | +| v3 | 0.5 / 0.5 | 4.74 | 5.52 | +0.12 (-0.06..+0.28) | 0.2 | 0 | +| v3 | 0.5 / 0.3 | 4.75 | 5.51 | +0.10 (-0.07..+0.28) | 0.0 | 0 | +| v3 | 0.3 / 0.3, only edges with >=0.5 s removed | 4.68 | 5.41 | +0.00 (-0.12..+0.12) | 0.1 | 0 | + +tune words: 13930 held-out words: 24990 (last detector) diff --git a/scripts/vad_bench/trim_regression/results/pooled.md b/scripts/vad_bench/trim_regression/results/pooled.md new file mode 100644 index 0000000..7b288dd --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/pooled.md @@ -0,0 +1,16 @@ +| comparison (B vs A) | set | words | WER A | WER B | delta (pp) | 95% CI | +|---|---|---|---|---|---|---| +| T0.3 vs T0 | noisy sinr 3 dets, tune+heldout | 49230 | 7.04 | 6.83 | -0.21 | -0.43..+0.03 | +| T0.3 vs T0 | talks 3 dets, all 9 | 64620 | 4.37 | 4.41 | +0.03 | -0.02..+0.09 | +| T0.5 vs T0.3 | noisy sinr 3 dets, tune+heldout | 49230 | 6.83 | 6.99 | +0.16 | -0.05..+0.36 | +| T0.5 vs T0.3 | talks 3 dets, all 9 | 64620 | 4.41 | 4.37 | -0.04 | -0.10..+0.01 | +| P0.5/0.3 vs T0.3 | noisy sinr 3 dets, tune+heldout | 49230 | 6.83 | 6.97 | +0.14 | -0.06..+0.33 | +| P0.5/0.3 vs T0.3 | talks 3 dets, all 9 | 64620 | 4.41 | 4.37 | -0.04 | -0.10..+0.01 | +| E0.5 vs T0.3 | noisy sinr 3 dets, tune+heldout | 49230 | 6.83 | 6.93 | +0.10 | -0.01..+0.21 | +| E0.5 vs T0.3 | talks 3 dets, all 9 | 64620 | 4.41 | 4.38 | -0.03 | -0.08..+0.01 | +| T0.5 vs T0 | noisy sinr 3 dets, tune+heldout | 49230 | 7.04 | 6.99 | -0.05 | -0.22..+0.11 | +| T0.5 vs T0 | talks 3 dets, all 9 | 64620 | 4.37 | 4.37 | -0.01 | -0.03..+0.02 | +| E0.5 vs T0 | noisy sinr 3 dets, tune+heldout | 49230 | 7.04 | 6.93 | -0.11 | -0.29..+0.10 | +| E0.5 vs T0 | talks 3 dets, all 9 | 64620 | 4.37 | 4.38 | +0.00 | -0.02..+0.02 | +| P0.5/0.3 vs T0 | noisy sinr 3 dets, tune+heldout | 49230 | 7.04 | 6.97 | -0.07 | -0.24..+0.09 | +| P0.5/0.3 vs T0 | talks 3 dets, all 9 | 64620 | 4.37 | 4.37 | -0.01 | -0.04..+0.02 | diff --git a/scripts/vad_bench/trim_regression/results/removed.txt b/scripts/vad_bench/trim_regression/results/removed.txt new file mode 100644 index 0000000..9b2a81d --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/removed.txt @@ -0,0 +1,9 @@ +ultra talk segments 299 mean len 24.7s | edges removed: >0 28.8% >=0.1s 16.1% >=0.5s 3.0% >=1s 0.0% >=2s 0.0% mean 0.05s p90 0.18s +ultra sinr-noisy segments 864 mean len 23.1s | edges removed: >0 66.3% >=0.1s 61.2% >=0.5s 37.2% >=1s 20.0% >=2s 0.0% mean 0.49s p90 1.38s +ultra sinr-clean segments 36 mean len 23.1s | edges removed: >0 72.2% >=0.1s 68.1% >=0.5s 40.3% >=1s 20.8% >=2s 0.0% mean 0.50s p90 1.37s +redux talk segments 296 mean len 25.0s | edges removed: >0 28.7% >=0.1s 18.4% >=0.5s 2.4% >=1s 0.0% >=2s 0.0% mean 0.05s p90 0.18s +redux sinr-noisy segments 860 mean len 23.2s | edges removed: >0 69.6% >=0.1s 60.6% >=0.5s 32.6% >=1s 15.3% >=2s 0.0% mean 0.42s p90 1.30s +redux sinr-clean segments 36 mean len 23.1s | edges removed: >0 75.0% >=0.1s 68.1% >=0.5s 40.3% >=1s 20.8% >=2s 0.0% mean 0.50s p90 1.36s +v3 talk segments 276 mean len 26.8s | edges removed: >0 22.3% >=0.1s 11.4% >=0.5s 1.8% >=1s 0.0% >=2s 0.0% mean 0.04s p90 0.15s +v3 sinr-noisy segments 864 mean len 23.1s | edges removed: >0 63.8% >=0.1s 59.3% >=0.5s 35.8% >=1s 22.3% >=2s 0.3% mean 0.52s p90 1.51s +v3 sinr-clean segments 36 mean len 23.1s | edges removed: >0 63.9% >=0.1s 59.7% >=0.5s 34.7% >=1s 20.8% >=2s 0.0% mean 0.49s p90 1.42s diff --git a/scripts/vad_bench/trim_regression/results/sdi.txt b/scripts/vad_bench/trim_regression/results/sdi.txt new file mode 100644 index 0000000..af64746 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/sdi.txt @@ -0,0 +1,7 @@ +S/D/I counts (T0 -> T0.3) +redux talks (9) words 21540 WER 4.80 -> 4.87 (+0.08; -0.03..+0.20) S 558->567 D 278->283 I 197->200 +redux noisy sinr white5+pink5+pink0 words 17946 WER 7.83 -> 7.71 (-0.12; -0.57..+0.45) S 1183->1154 D 132->133 I 90->97 +ultra talks (9) words 21540 WER 4.10 -> 4.11 (+0.01; -0.06..+0.08) S 450->452 D 238->239 I 196->195 +ultra noisy sinr white5+pink5+pink0 words 15642 WER 6.29 -> 6.02 (-0.27; -0.57..+0.01) S 801->758 D 128->125 I 55->58 +v3 talks (9) words 21540 WER 4.22 -> 4.23 (+0.01; -0.06..+0.10) S 440->436 D 237->244 I 233->232 +v3 noisy sinr white5+pink5+pink0 words 15642 WER 6.89 -> 6.62 (-0.26; -0.57..+0.02) S 815->796 D 199->177 I 63->63 diff --git a/scripts/vad_bench/trim_regression/results/segchange.txt b/scripts/vad_bench/trim_regression/results/segchange.txt new file mode 100644 index 0000000..f25d6e1 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/segchange.txt @@ -0,0 +1,14 @@ +redux talk T0->T0.3: segments 296, text changed 46 (15.5%); of those errors up 18, down 14, equal 14; net +17; sign test p=0.60 +redux sinr T0->T0.3: segments 860, text changed 439 (51.0%); of those errors up 139, down 194, equal 106; net -56; sign test p=0.00 +redux talk T0.3->T0.5: segments 296, text changed 49 (16.6%); of those errors up 14, down 21, equal 14; net -22; sign test p=0.31 +redux sinr T0.3->T0.5: segments 860, text changed 424 (49.3%); of those errors up 178, down 138, equal 108; net +85; sign test p=0.03 +ultra talk T0->T0.3: segments 299, text changed 30 (10.0%); of those errors up 10, down 15, equal 5; net +1; sign test p=0.42 +ultra sinr T0->T0.3: segments 291, text changed 149 (51.2%); of those errors up 43, down 68, equal 38; net -40; sign test p=0.02 +ultra talk T0.3->T0.5: segments 299, text changed 29 (9.7%); of those errors up 15, down 11, equal 3; net +0; sign test p=0.56 +ultra sinr T0.3->T0.5: segments 291, text changed 146 (50.2%); of those errors up 58, down 50, equal 38; net +19; sign test p=0.50 +v3 talk T0->T0.3: segments 276, text changed 21 (7.6%); of those errors up 9, down 9, equal 3; net +2; sign test p=1.00 +v3 sinr T0->T0.3: segments 279, text changed 151 (54.1%); of those errors up 43, down 66, equal 42; net -32; sign test p=0.03 +v3 talk T0.3->T0.5: segments 276, text changed 20 (7.2%); of those errors up 8, down 9, equal 3; net -3; sign test p=1.00 +v3 sinr T0.3->T0.5: segments 279, text changed 140 (50.2%); of those errors up 71, down 41, equal 28; net +48; sign test p=0.01 +redux talk T0.3->T0.32: segments 296, text changed 40 (13.5%); of those errors up 11, down 18, equal 11; net -11; sign test p=0.26 +redux talk T0.3->T0.35: segments 296, text changed 49 (16.6%); of those errors up 16, down 23, equal 10; net -19; sign test p=0.34 diff --git a/scripts/vad_bench/trim_regression/results/vad_cover.txt b/scripts/vad_bench/trim_regression/results/vad_cover.txt new file mode 100644 index 0000000..d61e078 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/vad_cover.txt @@ -0,0 +1,12 @@ +ultra talks correct words 20852 | outside the mask (any) 0.23% start-side>0.1s 0.01% >0.3s 0.00% end-side>0.1s 0.01% >0.3s 0.00% +ultra sinr clean correct words 1692 | outside the mask (any) 0.35% start-side>0.1s 0.06% >0.3s 0.00% end-side>0.1s 0.00% >0.3s 0.00% +ultra sinr noisy correct words 15265 | outside the mask (any) 1.32% start-side>0.1s 0.20% >0.3s 0.04% end-side>0.1s 0.24% >0.3s 0.07% +ultra sinr pink0+white0 correct words 4797 | outside the mask (any) 0.81% start-side>0.1s 0.15% >0.3s 0.02% end-side>0.1s 0.10% >0.3s 0.00% +redux talks correct words 20704 | outside the mask (any) 0.49% start-side>0.1s 0.03% >0.3s 0.00% end-side>0.1s 0.00% >0.3s 0.00% +redux sinr clean correct words 1948 | outside the mask (any) 1.13% start-side>0.1s 0.21% >0.3s 0.00% end-side>0.1s 0.00% >0.3s 0.00% +redux sinr noisy correct words 44859 | outside the mask (any) 1.15% start-side>0.1s 0.16% >0.3s 0.03% end-side>0.1s 0.00% >0.3s 0.00% +redux sinr pink0+white0 correct words 10397 | outside the mask (any) 0.92% start-side>0.1s 0.13% >0.3s 0.02% end-side>0.1s 0.02% >0.3s 0.00% +v3 talks correct words 20863 | outside the mask (any) 1.25% start-side>0.1s 0.15% >0.3s 0.00% end-side>0.1s 0.05% >0.3s 0.00% +v3 sinr clean correct words 1643 | outside the mask (any) 1.83% start-side>0.1s 0.18% >0.3s 0.06% end-side>0.1s 0.06% >0.3s 0.00% +v3 sinr noisy correct words 14628 | outside the mask (any) 4.03% start-side>0.1s 0.87% >0.3s 0.18% end-side>0.1s 0.40% >0.3s 0.12% +v3 sinr pink0+white0 correct words 4728 | outside the mask (any) 5.58% start-side>0.1s 1.52% >0.3s 0.34% end-side>0.1s 0.95% >0.3s 0.32% diff --git a/scripts/vad_bench/trim_regression/results/vad_edges.txt b/scripts/vad_bench/trim_regression/results/vad_edges.txt new file mode 100644 index 0000000..adbd2d6 --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/vad_edges.txt @@ -0,0 +1,149 @@ +Lateness of the speech start (VAD smoothed mask start minus true utterance start) and earliness of the end (true end minus VAD end). Positive = the VAD cuts into the word. +-- clean + ultra on_all n= 180 miss= 0.0% median= 98ms p90= 468 p95= 509 >0.3s=28.9% >0.5s= 6.7% + ultra on_edge n= 18 miss= 0.0% median= 129ms p90= 372 p95= 454 >0.3s=44.4% >0.5s= 0.0% + ultra off_all n= 180 miss= 0.0% median= -45ms p90= 356 p95= 383 >0.3s=18.9% >0.5s= 0.0% + ultra off_edge n= 22 miss= 0.0% median= 136ms p90= 350 p95= 365 >0.3s=18.2% >0.5s= 0.0% + redux on_all n= 180 miss= 0.0% median= 88ms p90= 473 p95= 514 >0.3s=26.7% >0.5s= 7.8% + redux on_edge n= 18 miss= 0.0% median= 209ms p90= 372 p95= 454 >0.3s=33.3% >0.5s= 0.0% + redux off_all n= 180 miss= 0.0% median= -47ms p90= 332 p95= 378 >0.3s=14.4% >0.5s= 0.0% + redux off_edge n= 24 miss= 0.0% median= 80ms p90= 350 p95= 364 >0.3s=16.7% >0.5s= 0.0% + v3 on_all n= 180 miss= 0.0% median= 110ms p90= 549 p95= 578 >0.3s=30.0% >0.5s=15.6% + v3 on_edge n= 10 miss= 0.0% median= 17ms p90= 336 p95= 336 >0.3s=20.0% >0.5s= 0.0% + v3 off_all n= 180 miss= 0.0% median= -5ms p90= 422 p95= 457 >0.3s=22.2% >0.5s= 1.1% + v3 off_edge n= 18 miss= 0.0% median= 108ms p90= 344 p95= 382 >0.3s=22.2% >0.5s= 0.0% +-- white20 + ultra on_all n= 270 miss= 0.0% median= 166ms p90= 514 p95= 546 >0.3s=31.1% >0.5s=11.5% + ultra on_edge n= 24 miss= 0.0% median= 230ms p90= 418 p95= 522 >0.3s=45.8% >0.5s= 8.3% + ultra off_all n= 270 miss= 0.0% median= 34ms p90= 385 p95= 424 >0.3s=23.3% >0.5s= 0.4% + ultra off_edge n= 30 miss= 0.0% median= 116ms p90= 302 p95= 350 >0.3s=10.0% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= 110ms p90= 471 p95= 518 >0.3s=29.3% >0.5s= 8.5% + redux on_edge n= 27 miss= 0.0% median= 209ms p90= 383 p95= 454 >0.3s=37.0% >0.5s= 0.0% + redux off_all n= 270 miss= 0.0% median= 9ms p90= 354 p95= 379 >0.3s=17.8% >0.5s= 1.1% + redux off_edge n= 34 miss= 0.0% median= 28ms p90= 270 p95= 296 >0.3s= 0.0% >0.5s= 0.0% + v3 on_all n= 270 miss= 0.0% median= 206ms p90= 549 p95= 569 >0.3s=36.7% >0.5s=15.6% + v3 on_edge n= 15 miss= 0.0% median= 49ms p90= 336 p95= 336 >0.3s=40.0% >0.5s= 0.0% + v3 off_all n= 270 miss= 0.0% median= 81ms p90= 457 p95= 466 >0.3s=25.6% >0.5s= 1.5% + v3 off_edge n= 27 miss= 0.0% median= 211ms p90= 350 p95= 382 >0.3s=14.8% >0.5s= 0.0% +-- white10 + ultra on_all n= 270 miss= 0.0% median= 154ms p90= 510 p95= 560 >0.3s=32.6% >0.5s=11.1% + ultra on_edge n= 22 miss= 0.0% median= 270ms p90= 441 p95= 454 >0.3s=36.4% >0.5s= 4.5% + ultra off_all n= 270 miss= 0.0% median= 102ms p90= 428 p95= 480 >0.3s=27.4% >0.5s= 4.4% + ultra off_edge n= 31 miss= 0.0% median= 161ms p90= 350 p95= 350 >0.3s=12.9% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= 79ms p90= 461 p95= 514 >0.3s=26.3% >0.5s= 7.8% + redux on_edge n= 28 miss= 0.0% median= 45ms p90= 349 p95= 454 >0.3s=17.9% >0.5s= 0.0% + redux off_all n= 270 miss= 0.0% median= 21ms p90= 333 p95= 387 >0.3s=14.8% >0.5s= 0.0% + redux off_edge n= 36 miss= 0.0% median= 95ms p90= 296 p95= 296 >0.3s= 2.8% >0.5s= 0.0% + v3 on_all n= 270 miss= 0.0% median= 211ms p90= 554 p95= 578 >0.3s=38.9% >0.5s=15.6% + v3 on_edge n= 16 miss= 0.0% median= 81ms p90= 337 p95= 400 >0.3s=43.8% >0.5s= 6.2% + v3 off_all n= 270 miss= 0.0% median= 89ms p90= 434 p95= 465 >0.3s=26.7% >0.5s= 1.9% + v3 off_edge n= 27 miss= 0.0% median= 241ms p90= 350 p95= 382 >0.3s=22.2% >0.5s= 0.0% +-- white5 + ultra on_all n= 270 miss= 0.0% median= 131ms p90= 510 p95= 560 >0.3s=32.2% >0.5s=10.4% + ultra on_edge n= 22 miss= 0.0% median= 238ms p90= 445 p95= 454 >0.3s=36.4% >0.5s= 4.5% + ultra off_all n= 270 miss= 0.0% median= 116ms p90= 465 p95= 525 >0.3s=30.4% >0.5s= 6.7% + ultra off_edge n= 32 miss= 0.0% median= 182ms p90= 445 p95= 456 >0.3s=21.9% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= 17ms p90= 445 p95= 509 >0.3s=22.2% >0.5s= 5.6% + redux on_edge n= 28 miss= 0.0% median= 45ms p90= 349 p95= 454 >0.3s=17.9% >0.5s= 3.6% + redux off_all n= 270 miss= 0.0% median= 7ms p90= 332 p95= 386 >0.3s=13.0% >0.5s= 0.4% + redux off_edge n= 36 miss= 0.0% median= 95ms p90= 296 p95= 314 >0.3s= 5.6% >0.5s= 0.0% + v3 on_all n= 270 miss= 0.0% median= 232ms p90= 569 p95= 590 >0.3s=40.4% >0.5s=16.7% + v3 on_edge n= 17 miss= 0.0% median= 81ms p90= 438 p95= 591 >0.3s=47.1% >0.5s=11.8% + v3 off_all n= 270 miss= 0.0% median= 108ms p90= 434 p95= 472 >0.3s=30.0% >0.5s= 4.1% + v3 off_edge n= 26 miss= 0.0% median= 226ms p90= 350 p95= 374 >0.3s=30.8% >0.5s= 0.0% +-- white0 + ultra on_all n= 270 miss= 0.0% median= 4ms p90= 454 p95= 510 >0.3s=27.8% >0.5s= 5.9% + ultra on_edge n= 27 miss= 0.0% median= 25ms p90= 403 p95= 454 >0.3s=37.0% >0.5s= 0.0% + ultra off_all n= 270 miss= 0.0% median= 49ms p90= 465 p95= 523 >0.3s=26.3% >0.5s= 7.4% + ultra off_edge n= 32 miss= 0.0% median= 109ms p90= 374 p95= 456 >0.3s=18.8% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= -51ms p90= 405 p95= 464 >0.3s=18.5% >0.5s= 3.7% + redux on_edge n= 30 miss= 0.0% median= -55ms p90= 348 p95= 452 >0.3s=16.7% >0.5s= 0.0% + redux off_all n= 270 miss= 0.0% median= -95ms p90= 355 p95= 389 >0.3s=14.4% >0.5s= 1.5% + redux off_edge n= 37 miss= 0.0% median= 7ms p90= 296 p95= 310 >0.3s= 5.4% >0.5s= 0.0% + v3 on_all n= 270 miss= 1.5% median= 251ms p90= 586 p95= 642 >0.3s=44.4% >0.5s=18.0% + v3 on_edge n= 15 miss= 0.0% median= 81ms p90= 497 p95= 525 >0.3s=46.7% >0.5s= 6.7% + v3 off_all n= 270 miss= 1.5% median= 150ms p90= 470 p95= 530 >0.3s=33.5% >0.5s= 6.8% + v3 off_edge n= 24 miss= 0.0% median= 241ms p90= 392 p95= 392 >0.3s=37.5% >0.5s= 0.0% +-- pink20 + ultra on_all n= 270 miss= 0.0% median= 154ms p90= 511 p95= 536 >0.3s=31.1% >0.5s=11.5% + ultra on_edge n= 24 miss= 0.0% median= 256ms p90= 418 p95= 522 >0.3s=50.0% >0.5s= 8.3% + ultra off_all n= 270 miss= 0.0% median= 30ms p90= 378 p95= 409 >0.3s=20.7% >0.5s= 0.4% + ultra off_edge n= 30 miss= 0.0% median= 104ms p90= 302 p95= 350 >0.3s=10.0% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= 98ms p90= 490 p95= 521 >0.3s=28.1% >0.5s= 9.3% + redux on_edge n= 26 miss= 0.0% median= 209ms p90= 395 p95= 454 >0.3s=38.5% >0.5s= 0.0% + redux off_all n= 270 miss= 0.0% median= -24ms p90= 330 p95= 378 >0.3s=15.9% >0.5s= 1.1% + redux off_edge n= 35 miss= 0.0% median= 28ms p90= 296 p95= 317 >0.3s= 5.7% >0.5s= 0.0% + v3 on_all n= 270 miss= 0.0% median= 185ms p90= 555 p95= 578 >0.3s=38.5% >0.5s=15.6% + v3 on_edge n= 15 miss= 0.0% median= 25ms p90= 336 p95= 336 >0.3s=40.0% >0.5s= 0.0% + v3 off_all n= 270 miss= 0.0% median= 82ms p90= 430 p95= 464 >0.3s=27.4% >0.5s= 2.2% + v3 off_edge n= 27 miss= 0.0% median= 200ms p90= 350 p95= 382 >0.3s=22.2% >0.5s= 0.0% +-- pink10 + ultra on_all n= 270 miss= 0.0% median= 154ms p90= 514 p95= 543 >0.3s=30.7% >0.5s=11.9% + ultra on_edge n= 23 miss= 0.0% median= 209ms p90= 429 p95= 526 >0.3s=34.8% >0.5s= 8.7% + ultra off_all n= 270 miss= 0.0% median= 49ms p90= 408 p95= 470 >0.3s=24.4% >0.5s= 3.3% + ultra off_edge n= 32 miss= 0.0% median= 156ms p90= 296 p95= 350 >0.3s= 9.4% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= 55ms p90= 458 p95= 510 >0.3s=24.8% >0.5s= 7.0% + redux on_edge n= 27 miss= 0.0% median= 252ms p90= 454 p95= 454 >0.3s=29.6% >0.5s= 3.7% + redux off_all n= 270 miss= 0.0% median= -35ms p90= 329 p95= 373 >0.3s=12.6% >0.5s= 0.0% + redux off_edge n= 35 miss= 0.0% median= 28ms p90= 286 p95= 296 >0.3s= 2.9% >0.5s= 0.0% + v3 on_all n= 270 miss= 0.0% median= 206ms p90= 569 p95= 590 >0.3s=38.1% >0.5s=15.9% + v3 on_edge n= 15 miss= 0.0% median= 81ms p90= 336 p95= 336 >0.3s=40.0% >0.5s= 0.0% + v3 off_all n= 270 miss= 0.0% median= 94ms p90= 442 p95= 465 >0.3s=28.5% >0.5s= 2.6% + v3 off_edge n= 27 miss= 0.0% median= 241ms p90= 382 p95= 389 >0.3s=22.2% >0.5s= 0.0% +-- pink5 + ultra on_all n= 270 miss= 0.0% median= 128ms p90= 510 p95= 541 >0.3s=31.1% >0.5s=10.7% + ultra on_edge n= 23 miss= 0.0% median= 209ms p90= 429 p95= 526 >0.3s=34.8% >0.5s= 8.7% + ultra off_all n= 270 miss= 0.0% median= 91ms p90= 409 p95= 466 >0.3s=25.2% >0.5s= 2.6% + ultra off_edge n= 32 miss= 0.0% median= 152ms p90= 296 p95= 320 >0.3s= 6.2% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= -23ms p90= 453 p95= 510 >0.3s=21.9% >0.5s= 6.7% + redux on_edge n= 24 miss= 0.0% median= 238ms p90= 418 p95= 518 >0.3s=25.0% >0.5s= 8.3% + redux off_all n= 270 miss= 0.0% median= -57ms p90= 330 p95= 373 >0.3s=12.6% >0.5s= 0.4% + redux off_edge n= 36 miss= 0.0% median= 7ms p90= 283 p95= 296 >0.3s= 2.8% >0.5s= 0.0% + v3 on_all n= 270 miss= 0.0% median= 227ms p90= 586 p95= 610 >0.3s=40.4% >0.5s=16.3% + v3 on_edge n= 15 miss= 0.0% median= 81ms p90= 337 p95= 337 >0.3s=40.0% >0.5s= 0.0% + v3 off_all n= 270 miss= 0.0% median= 107ms p90= 434 p95= 468 >0.3s=27.0% >0.5s= 3.3% + v3 off_edge n= 25 miss= 0.0% median= 241ms p90= 388 p95= 392 >0.3s=24.0% >0.5s= 0.0% +-- pink0 + ultra on_all n= 270 miss= 0.0% median= -76ms p90= 391 p95= 456 >0.3s=18.5% >0.5s= 3.0% + ultra on_edge n= 28 miss= 0.0% median= -47ms p90= 372 p95= 454 >0.3s=25.0% >0.5s= 3.6% + ultra off_all n= 270 miss= 0.0% median= -42ms p90= 378 p95= 424 >0.3s=15.2% >0.5s= 1.1% + ultra off_edge n= 34 miss= 0.0% median= 72ms p90= 295 p95= 315 >0.3s= 5.9% >0.5s= 0.0% + redux on_all n= 270 miss= 0.0% median= -74ms p90= 356 p95= 453 >0.3s=14.8% >0.5s= 2.6% + redux on_edge n= 25 miss= 0.0% median= -64ms p90= 300 p95= 330 >0.3s=12.0% >0.5s= 0.0% + redux off_all n= 270 miss= 0.0% median= -147ms p90= 303 p95= 379 >0.3s=10.4% >0.5s= 0.7% + redux off_edge n= 35 miss= 0.0% median= -168ms p90= 270 p95= 296 >0.3s= 2.9% >0.5s= 0.0% + v3 on_all n= 270 miss= 0.0% median= 264ms p90= 588 p95= 672 >0.3s=45.6% >0.5s=18.9% + v3 on_edge n= 15 miss= 0.0% median= 81ms p90= 337 p95= 337 >0.3s=40.0% >0.5s= 0.0% + v3 off_all n= 270 miss= 0.0% median= 120ms p90= 490 p95= 583 >0.3s=31.9% >0.5s= 9.6% + v3 off_edge n= 24 miss= 0.0% median= 254ms p90= 392 p95= 392 >0.3s=29.2% >0.5s= 4.2% +-- ALL-noisy + ultra on_all n=2160 miss= 0.0% median= 79ms p90= 490 p95= 537 >0.3s=29.4% >0.5s= 9.5% + ultra on_edge n= 193 miss= 0.0% median= 209ms p90= 454 p95= 531 >0.3s=37.3% >0.5s= 5.7% + ultra off_all n=2160 miss= 0.0% median= 48ms p90= 409 p95= 469 >0.3s=24.1% >0.5s= 3.3% + ultra off_edge n= 253 miss= 0.0% median= 136ms p90= 350 p95= 350 >0.3s=11.9% >0.5s= 0.0% + redux on_all n=2160 miss= 0.0% median= -5ms p90= 454 p95= 510 >0.3s=23.2% >0.5s= 6.4% + redux on_edge n= 215 miss= 0.0% median= 58ms p90= 418 p95= 454 >0.3s=24.2% >0.5s= 1.9% + redux off_all n=2160 miss= 0.0% median= -45ms p90= 331 p95= 383 >0.3s=13.9% >0.5s= 0.6% + redux off_edge n= 284 miss= 0.0% median= 28ms p90= 296 p95= 296 >0.3s= 3.5% >0.5s= 0.0% + v3 on_all n=2160 miss= 0.2% median= 237ms p90= 569 p95= 590 >0.3s=40.4% >0.5s=16.6% + v3 on_edge n= 123 miss= 0.0% median= 81ms p90= 337 p95= 366 >0.3s=42.3% >0.5s= 3.3% + v3 off_all n=2160 miss= 0.2% median= 106ms p90= 442 p95= 482 >0.3s=28.8% >0.5s= 4.0% + v3 off_edge n= 207 miss= 0.0% median= 232ms p90= 382 p95= 392 >0.3s=25.1% >0.5s= 0.5% + +Fraction of true onsets (offsets) at which the RAW frame (p>=0.5) at time t relative to the true onset (end) is speech. t in s. +-- clean onset t=-0.48 -0.32 -0.16 +0.00 +0.16 +0.32 +0.48 +0.64 +0.80 +0.96 + ultra 11 8 2 38 54 70 90 99 98 98 + redux 10 8 0 40 53 71 90 99 98 99 + v3 10 4 0 7 48 68 81 99 98 97 +-- clean offset t=-0.48 -0.32 -0.16 +0.00 +0.16 +0.32 +0.48 +0.64 +0.80 +0.96 + ultra 100 83 67 56 28 2 7 14 16 33 + redux 100 84 67 49 21 3 8 14 17 35 + v3 97 77 58 47 1 0 6 10 16 26 +-- ALL-noisy onset t=-0.48 -0.32 -0.16 +0.00 +0.16 +0.32 +0.48 +0.64 +0.80 +0.96 + ultra 11 7 3 36 49 71 87 97 98 98 + redux 15 10 5 41 52 74 91 99 98 99 + v3 9 4 0 5 41 60 78 95 97 94 +-- ALL-noisy offset t=-0.48 -0.32 -0.16 +0.00 +0.16 +0.32 +0.48 +0.64 +0.80 +0.96 + ultra 94 74 58 34 11 5 8 14 19 34 + redux 99 85 65 45 19 8 11 18 22 38 + v3 92 72 55 33 1 0 3 9 15 25 diff --git a/scripts/vad_bench/trim_regression/results/validate_cli_vs_harness.txt b/scripts/vad_bench/trim_regression/results/validate_cli_vs_harness.txt new file mode 100644 index 0000000..c5f081a --- /dev/null +++ b/scripts/vad_bench/trim_regression/results/validate_cli_vs_harness.txt @@ -0,0 +1,33 @@ +ultra talk_GaryFlake-merged 0.3 segs 16 16 bounds_match True text_match True +ultra talk_GaryFlake-merged 0.0 segs 16 16 bounds_match True text_match True +ultra sn_heldout_pink0_L00s0 0.3 segs 3 3 bounds_match True text_match True +ultra sn_heldout_pink0_L00s0 0.0 segs 3 3 bounds_match True text_match True +ultra sn_heldout_white5_L03s1 0.3 segs 3 3 bounds_match True text_match True +ultra sn_heldout_white5_L03s1 0.0 segs 3 3 bounds_match True text_match True +ultra sn_tune_clean_L02s0 0.3 segs 3 3 bounds_match True text_match True +ultra sn_tune_clean_L02s0 0.0 segs 3 3 bounds_match True text_match True +ultra ins_GaryFlake-merged_pink-5 0.3 segs 6 6 bounds_match True text_match True +ultra ins_GaryFlake-merged_pink-5 0.0 segs 6 6 bounds_match True text_match True +redux talk_GaryFlake-merged 0.3 segs 16 16 bounds_match True text_match True +redux talk_GaryFlake-merged 0.0 segs 16 16 bounds_match True text_match True +redux sn_heldout_pink0_L00s0 0.3 segs 3 3 bounds_match True text_match True +redux sn_heldout_pink0_L00s0 0.0 segs 3 3 bounds_match True text_match True +redux sn_heldout_white5_L03s1 0.3 segs 3 3 bounds_match True text_match True +redux sn_heldout_white5_L03s1 0.0 segs 3 3 bounds_match True text_match True +redux sn_tune_clean_L02s0 0.3 segs 3 3 bounds_match True text_match True +redux sn_tune_clean_L02s0 0.0 segs 3 3 bounds_match True text_match True +redux ins_GaryFlake-merged_pink-5 0.3 segs 7 7 bounds_match True text_match True +redux ins_GaryFlake-merged_pink-5 0.0 segs 7 7 bounds_match True text_match True +v3 talk_GaryFlake-merged 0.3 segs 14 14 bounds_match True text_match True +v3 talk_GaryFlake-merged 0.0 segs 14 14 bounds_match True text_match True +v3 sn_heldout_pink0_L00s0 0.3 segs 3 3 bounds_match True text_match True +v3 sn_heldout_pink0_L00s0 0.0 segs 3 3 bounds_match True text_match True +v3 sn_heldout_white5_L03s1 0.3 segs 3 3 bounds_match True text_match True +v3 sn_heldout_white5_L03s1 0.0 segs 3 3 bounds_match True text_match True +v3 sn_tune_clean_L02s0 0.3 segs 3 3 bounds_match True text_match True +v3 sn_tune_clean_L02s0 0.0 segs 3 3 bounds_match True text_match True +v3 ins_GaryFlake-merged_pink-5 0.3 segs 5 5 bounds_match True text_match True +v3 ins_GaryFlake-merged_pink-5 0.0 segs 5 5 bounds_match True text_match True +segment mismatches 0 text mismatches 0 of 30 + +[exited with code 0] diff --git a/scripts/vad_bench/trim_regression/scripts/audio_clip.py b/scripts/vad_bench/trim_regression/scripts/audio_clip.py new file mode 100644 index 0000000..0ab2146 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/audio_clip.py @@ -0,0 +1,21 @@ +"""Against the construction truth of the sinr files (utterance spans = first/last 10 ms frame above 1% of the peak RMS): +how much of an utterance does a trimmed segment cut off at its start/end? Segment edges next to an utterance only.""" +import sys; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +vs = sys.argv[1].split(","); dets = sys.argv[2].split(",") if len(sys.argv) > 2 else list(DET) +nm = [n for n in names("sinr", None, W_ + P_ + ["clean"])] +for det in dets: + print("==", det) + for v in vs: + cs = []; ce = [] + for n in nm: + m = MAN[n]; s0 = segs_for(det, n, "T0"); sb = segs_for(det, n, v) + for (a0, b0), (a, b) in zip(s0, sb): + if not any(us < a0 < ue for us, ue in m["spans"]): # the start cut lies in a gap + nxt = [us for us, ue in m["spans"] if us >= a0 - 1e-6 and us < b0] + if nxt: cs.append(max(0.0, a - nxt[0])) + if not any(us < b0 < ue for us, ue in m["spans"]): # the end cut lies in a gap + prv = [ue for us, ue in m["spans"] if ue <= b0 + 1e-6 and ue > a0] + if prv: ce.append(max(0.0, prv[-1] - b)) + cs = np.array(cs); ce = np.array(ce) + print(f" {v:8s} segment starts: cut into the first utterance {100*np.mean(cs>0.0):4.1f}% (>=50ms {100*np.mean(cs>=0.05):4.1f}%, >=100ms {100*np.mean(cs>=0.1):4.1f}%, >=300ms {100*np.mean(cs>=0.3):4.1f}%, max {cs.max()*1000:.0f}ms) | ends: cut into the last utterance {100*np.mean(ce>0.0):4.1f}% (>=100ms {100*np.mean(ce>=0.1):4.1f}%, >=300ms {100*np.mean(ce>=0.3):4.1f}%, max {ce.max()*1000:.0f}ms) n={len(cs)}") diff --git a/scripts/vad_bench/trim_regression/scripts/boundary.py b/scripts/vad_bench/trim_regression/scripts/boundary.py new file mode 100644 index 0000000..deba6c4 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/boundary.py @@ -0,0 +1,57 @@ +"""Where do the errors change between two variants? usage: boundary.py DET VA VB [split] [kinds] +Zones are defined from VB's trimmed segments (the 'narrow' ones): cutaway = inside VA's segment but outside VB's; +start/end edge = first/last EDGE s inside VB's segment; interior = the rest. Time of an error: word mid for S/I, +for D the position inside the gap between the neighbouring hypothesis words.""" +import sys, collections; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +EDGE = 0.5 +def events_with_time(det, name, vn): + r = file_errors(det, name, vn); segs = segs_for(det, name, vn); hy = hyp(det, name, segs) + ev = r["events"]; out = [] + # hypothesis neighbours for deletions: need the full alignment order -> recompute positions from the event list + full = align(norm(MAN[name]["ref"]), hy) + prev_end = 0.0; pend = []; hmax = None + def flush(next_start): + nonlocal pend + k = len(pend) + for i, (j) in enumerate(pend): + out.append(("D", prev_end + (next_start - prev_end) * (i + 1) / (k + 1), j, None)) + pend = [] + for kind, rj, hj in full: + if kind == "D": pend.append(rj); continue + t0, t1 = hy[hj][1], hy[hj][2] + if pend: flush(t0) + if kind in "SI": out.append((kind, (t0 + t1) / 2, rj, hj)) + prev_end = t1 + if pend: flush(prev_end + 0.5) + return out, hy, full, segs +def zone(t, segsB): + for i, (s, e) in enumerate(segsB): + if s <= t <= e: + if t < s + EDGE: return "start-edge" + if t > e - EDGE: return "end-edge" + return "interior" + return "cutaway" +if __name__ == "__main__": + det, va, vb = sys.argv[1:4]; split = sys.argv[4] if len(sys.argv) > 4 else "all"; kinds = sys.argv[5].split(",") if len(sys.argv) > 5 else ["talk", "sinr"] + nms = [n for k in kinds for n in names(k, None if split == "all" else split)] + tab = collections.defaultdict(lambda: [0, 0]); clipped = collections.Counter() + for nm in nms: + try: ea, hya, fa, sa = events_with_time(det, nm, va); eb, hyb, fb, sb = events_with_time(det, nm, vb) + except KeyError: continue + for kind, t, rj, hj in ea: tab[(zone(t, sb), kind)][0] += 1 + for kind, t, rj, hj in eb: tab[(zone(t, sb), kind)][1] += 1 + # status of VA hypothesis tokens that VB's segments cut away + stat = {hj: kind for kind, rj, hj in fa if hj is not None} + for hj, (tok, st, en, si, cf) in enumerate(hya): + mid = (st + en) / 2 + if zone(mid, sb) == "cutaway": clipped["tokens in cut-away"] += 1; clipped[" of which correct(C)" if stat[hj] == "C" else " of which error (S/I)"] += 1 + clipped["tokens total (A)"] += len(hya) + print(f"== {det} {va} -> {vb} split={split} kinds={kinds} (errors by zone: A / B / B-A)") + for z in ("cutaway", "start-edge", "end-edge", "interior"): + row = [] + for k in "SDI": + a, b = tab[(z, k)]; row.append(f"{k}: {a:4d}/{b:4d}/{b-a:+4d}") + a = sum(tab[(z, k)][0] for k in "SDI"); b = sum(tab[(z, k)][1] for k in "SDI") + print(f" {z:11s} " + " ".join(row) + f" total {a:4d}/{b:4d}/{b-a:+4d}") + print(" ", dict(clipped)) diff --git a/scripts/vad_bench/trim_regression/scripts/clip_count.py b/scripts/vad_bench/trim_regression/scripts/clip_count.py new file mode 100644 index 0000000..b1c7158 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/clip_count.py @@ -0,0 +1,27 @@ +"""Words of the untrimmed (T0) decode that a variant's segments cut: counted from the T0 hypothesis and its alignment to the reference. +cut = the word interval overlaps the removed part by more than half (mid outside the kept segment) ; touched = any overlap with the removed part.""" +import sys, collections; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +vs = sys.argv[1].split(","); dets = sys.argv[2].split(",") if len(sys.argv) > 2 else list(DET) +sets = (("talks", names("talk")), ("sinr noisy", names("sinr", None, W_ + P_)), ("sinr clean", names("sinr", None, ["clean"]))) +for det in dets: + print(f"== {det}") + for sname, nm in sets: + base_tok = 0; rows = {v: collections.Counter() for v in vs} + for n in nm: + try: + s0 = segs_for(det, n, "T0"); hy = hyp(det, n, s0); ev = align(norm(MAN[n]["ref"]), hy) + except KeyError: continue + st = {hj: k for k, rj, hj in ev if hj is not None}; base_tok += len(hy) + for v in vs: + sb = segs_for(det, n, v) + for hj, (tok, s, e, si, cf) in enumerate(hy): + a, b = sb[si] # same segment index (trim keeps the segment list) + out = max(0.0, a - s) + max(0.0, e - b) + if out <= 0: continue + kind = "correct" if st[hj] == "C" else "error" + rows[v]["touched " + kind] += 1 + if out > 0.5 * (e - s): rows[v]["cut " + kind] += 1 + print(f" {sname:11s} T0 words {base_tok}") + for v in vs: + c = rows[v]; print(f" {v:12s} cut correct {c['cut correct']:4d} cut error {c['cut error']:4d} | touched correct {c['touched correct']:4d} touched error {c['touched error']:4d}") diff --git a/scripts/vad_bench/trim_regression/scripts/common.py b/scripts/vad_bench/trim_regression/scripts/common.py new file mode 100644 index 0000000..7507dd5 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/common.py @@ -0,0 +1,19 @@ +import json, os, re +import numpy as np +# Paths come from the environment. PK_WORK: work directory (data/, probs/, cache/, tmp/ live there; default +# is the current directory). PK_CLI: the patched parakeet-cli. PK_GGUF: directory with the model files. +W = os.path.abspath(os.environ.get("PK_WORK", os.getcwd())) +CLI = os.environ.get("PK_CLI", f"{W}/build/examples/cli/parakeet-cli") +GG = os.environ.get("PK_GGUF", f"{W}/gguf") +os.makedirs(f"{W}/tmp", exist_ok=True) +DET = { # name: (asr model, silero or None) + "ultra": (f"{GG}/ultra-q8_0.gguf", None), + "redux": (f"{GG}/redux-keep.gguf", None), + "v3": (f"{GG}/tdt-0.6b-v3-q8_0.gguf", f"{GG}/silero-vad-f16.gguf"), +} +HEAD = dict(threshold=0.5, frame_sec=0.08, max_seg_sec=30.0, min_pause_sec=0.2, min_seg_sec=1.0, bridge_sec=0.1, min_speech_sec=0.1) +SIL = dict(HEAD, frame_sec=0.032, min_speech_sec=0.25, min_pause_sec=0.1, bridge_sec=0.1) +OPTS = {"ultra": HEAD, "redux": HEAD, "v3": SIL} +def manifest(): return json.load(open(f"{W}/data/manifest.json")) +def probs(det, name): return np.fromfile(f"{W}/probs/{det}/{name}.f32", dtype=np.float32) +def norm(s): return re.sub(r"[^a-z0-9' ]", " ", s.lower().replace("-", " ")).split() diff --git a/scripts/vad_bench/trim_regression/scripts/dec.py b/scripts/vad_bench/trim_regression/scripts/dec.py new file mode 100644 index 0000000..8f07493 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/dec.py @@ -0,0 +1,34 @@ +"""Decode segments through the (patched) CLI with a per-(det,file) cache of segment -> words.""" +import json, os, subprocess, sys, tempfile +sys.path.insert(0, os.path.dirname(__file__)); from common import * +def key(s, e): return f"{s:.4f},{e:.4f}" +def cpath(det, name): return f"{W}/cache/{det}/{name}.json" +def load_cache(det, name): + f = cpath(det, name) + return json.load(open(f)) if os.path.exists(f) else {} +def run_segments(det, name, segs, threads=5, tmpdir=None): + cache = load_cache(det, name) + miss = sorted({(round(s, 4), round(e, 4)) for s, e in segs if key(s, e) not in cache}) + if not miss: return cache + model, sil = DET[det] + with tempfile.TemporaryDirectory(dir=f"{W}/tmp") as td: + sf, so = f"{td}/seg.txt", f"{td}/out.txt" + open(sf, "w").write("".join(f"{s:.4f} {e:.4f}\n" for s, e in miss)) + cmd = [CLI, "transcribe", "--model", model, "--input", f"{W}/data/{name}.wav", "--json", "--threads", str(threads), "--vad"] + (["--vad-model", sil] if sil else []) + r = subprocess.run(cmd, capture_output=True, text=True, env=dict(os.environ, PK_SEGMENTS=sf, PK_SEGOUT=so)) + if r.returncode != 0: raise RuntimeError(r.stderr[-300:]) + lines = open(so).read().split("\n") if os.path.exists(so) else [] + lines = [l for l in lines if l] + # one line per slice, in the order of the segment file; slices of no speech never happen here (all kept) + assert len(lines) == len(miss), (len(lines), len(miss), name) + for (s, e), l in zip(miss, lines): + a, b, w = (l.split("\t") + [""])[:3] + words = [] + import re as _re + for tok in _re.split(r" (?=-?\d+\.\d{3}\|-?\d+\.\d{3}\|\d\.\d{3}\|)", w): + if tok: + st, en, cf, tx = tok.split("|", 3); words.append([float(st), float(en), float(cf), tx]) + cache[key(s, e)] = dict(out=[float(a), float(b)], words=words) + os.makedirs(os.path.dirname(cpath(det, name)), exist_ok=True) + json.dump(cache, open(cpath(det, name) + ".tmp", "w")); os.rename(cpath(det, name) + ".tmp", cpath(det, name)) + return cache diff --git a/scripts/vad_bench/trim_regression/scripts/fetch.py b/scripts/vad_bench/trim_regression/scripts/fetch.py new file mode 100644 index 0000000..a89d0be --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/fetch.py @@ -0,0 +1,26 @@ +#!/usr/bin/env python3 +"""Stream TED-LIUM long-form test talks (<=1500 s, all of them) and LibriSpeech test-clean (every 13th utt, with speaker ids).""" +import io, json, os, re, sys +import numpy as np, soundfile as sf, librosa +from datasets import Audio, load_dataset +out = sys.argv[1]; os.makedirs(out, exist_ok=True) +os.environ["HF_HOME"] = os.path.abspath(os.path.join(out, "..", "hfhome")) +ds = load_dataset("distil-whisper/tedlium-long-form", split="test", streaming=True).cast_column("audio", Audio(decode=False)) +for i, ex in enumerate(ds): + y, sr = sf.read(io.BytesIO(ex["audio"]["bytes"]), dtype="float32") + if y.ndim > 1: y = y.mean(axis=1) + if sr != 16000: y = librosa.resample(y, orig_sr=sr, target_sr=16000) + name = re.sub(r"[^A-Za-z0-9_-]", "", os.path.basename(ex["audio"]["path"] or f"talk{i}").rsplit(".", 1)[0]) or f"talk{i}" + print("talk", i, name, round(len(y) / 16000), flush=True) + if len(y) / 16000 > 1500: continue + sf.write(f"{out}/talk_{name}.wav", y, 16000, subtype="PCM_16") + open(f"{out}/talk_{name}.txt", "w").write(" ".join(re.sub(r"<[^>]*>", " ", ex["text"]).split()) + "\n") +ds = load_dataset("openslr/librispeech_asr", "clean", split="test", streaming=True).cast_column("audio", Audio(decode=False)) +U, T, SP = [], [], [] +for i, ex in enumerate(ds): + if i % 13: continue + y, sr = sf.read(io.BytesIO(ex["audio"]["bytes"]), dtype="int16"); assert sr == 16000 + U.append(y); T.append(ex["text"].lower()); SP.append(int(ex["speaker_id"])) +np.save(f"{out}/libri.npy", np.array(U, dtype=object), allow_pickle=True) +json.dump({"text": T, "speaker": SP}, open(f"{out}/libri.json", "w")) +print("libri", len(U), len(set(SP)), flush=True) diff --git a/scripts/vad_bench/trim_regression/scripts/gens.py b/scripts/vad_bench/trim_regression/scripts/gens.py new file mode 100644 index 0000000..da45e90 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/gens.py @@ -0,0 +1,29 @@ +import numpy as np +SR=16000 +def rms(x): return float(np.sqrt((x ** 2).mean()) + 1e-12) +def at_db(x, db): return x * (10 ** (db / 20) / rms(x)) +def pink(n, r): + X = np.fft.rfft(r.standard_normal(n)); f = np.arange(len(X)); f[0] = 1 + y = np.fft.irfft(X / np.sqrt(f), n); return y / y.std() +def clicks(n, r): + x = np.zeros(n); t = 0 + while t < n - 400: + k = np.arange(400); x[t:t + 400] += r.uniform(0.3, 1) * r.choice([-1, 1]) * np.exp(-k / 40) * np.sin(k * 0.8) + t += int(r.uniform(0.15, 0.5) * SR) + return x +def music(n, r): + t = np.arange(n) / SR; x = np.zeros(n) + for f0 in (220.0, 261.63, 329.63, 440.0): + f = f0 * (1 + 0.003 * np.sin(2 * np.pi * r.uniform(4, 6) * t + r.uniform(0, 6))) + ph = 2 * np.pi * np.cumsum(f) / SR + for h, a in ((1, 1.0), (2, 0.5), (3, 0.25)): x += a * np.sin(h * ph) + env = 0.6 + 0.4 * np.sin(2 * np.pi * 0.5 * t + r.uniform(0, 6)) ** 2 + return x * env +def hum(n, r): + t = np.arange(n) / SR; f = r.choice([50.0, 60.0]) + return sum(np.sin(2 * np.pi * f * h * t + r.uniform(0, 6)) / h for h in (1, 2, 3, 5)) +def tone(n, r): return np.sin(2 * np.pi * float(r.choice([440, 1000, 2000])) * np.arange(n) / SR) +def sweep(n, r): + t = np.arange(n) / SR; T = t[-1]; return np.sin(2 * np.pi * (50 * t + (7500 - 50) * t * t / (2 * T))) +GEN = {"white": lambda n, r: r.standard_normal(n), "pink": pink, "clicks": clicks, "music": music, + "hum": hum, "tone": tone, "sweep": sweep} diff --git a/scripts/vad_bench/trim_regression/scripts/groups.py b/scripts/vad_bench/trim_regression/scripts/groups.py new file mode 100644 index 0000000..1d54bdb --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/groups.py @@ -0,0 +1,27 @@ +from lib import * +def names(kind=None, split=None, conds=None, fam=None): + out = [] + for m in MAN.values(): + if kind and m["kind"] != kind: continue + if m.get("dur", 99) <= 30: continue + if split and m["split"] != split: continue + if kind == "sinr": + c = m["cond"] + if conds and c not in conds: continue + if fam == "noisy" and c == "clean": continue + out.append(m["name"]) + return out +W_ = ["white20", "white10", "white5", "white0"]; P_ = ["pink20", "pink10", "pink5", "pink0"] +def G(split): + s = None if split == "all" else split + return { + f"talks[{split}]": names("talk", s), + f"clean[{split}]": names("sinr", s, ["clean"]), + f"white[{split}]": names("sinr", s, W_), + f"pink[{split}]": names("sinr", s, P_), + f"white0[{split}]": names("sinr", s, ["white0"]), + f"pink0[{split}]": names("sinr", s, ["pink0"]), + f"pink5[{split}]": names("sinr", s, ["pink5"]), + f"noisy-all[{split}]": names("sinr", s, W_ + P_), + f"sinr-all[{split}]": names("sinr", s, ["clean"] + W_ + P_), + } diff --git a/scripts/vad_bench/trim_regression/scripts/lib.py b/scripts/vad_bench/trim_regression/scripts/lib.py new file mode 100644 index 0000000..edd1c79 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/lib.py @@ -0,0 +1,90 @@ +import os, sys, pickle, json +import numpy as np, soundfile as sf, jiwer +sys.path.insert(0, os.path.dirname(__file__)); from common import *; from seg import *; import dec, variants +MAN = {m["name"]: m for m in manifest()} +def total_sec(name): + m = MAN[name] + if "dur" in m and m["kind"] != "insert": return m["dur"] + return sf.info(f"{W}/data/{name}.wav").frames / 16000 +def segs_for(det, name, vn): return segments(probs(det, name), total_sec(name), OPTS[det], variants.get(vn, det)) +def hyp(det, name, segs): + """tokens: list of (token, start, end, seg_idx, conf)""" + cache = dec.load_cache(det, name); out = [] + for i, (s, e) in enumerate(segs): + c = cache[dec.key(s, e)] + for w in c["words"]: + for t in norm(w[3]): out.append((t, w[0], w[1], i, w[2])) + return out +def align(ref, hyp_tokens): + """returns events: list of (kind, ref_idx or None, hyp_idx or None), kind in C,S,D,I""" + h = [t[0] for t in hyp_tokens] + if not h: return [("D", j, None) for j in range(len(ref))] + r = jiwer.process_words(" ".join(ref), " ".join(h)) + ev = [] + for ch in r.alignments[0]: + if ch.type == "equal": + ev += [("C", ch.ref_start_idx + k, ch.hyp_start_idx + k) for k in range(ch.ref_end_idx - ch.ref_start_idx)] + elif ch.type == "substitute": + ev += [("S", ch.ref_start_idx + k, ch.hyp_start_idx + k) for k in range(ch.ref_end_idx - ch.ref_start_idx)] + elif ch.type == "delete": + ev += [("D", ch.ref_start_idx + k, None) for k in range(ch.ref_end_idx - ch.ref_start_idx)] + else: + ev += [("I", None, ch.hyp_start_idx + k) for k in range(ch.hyp_end_idx - ch.hyp_start_idx)] + return ev +def utt_index(m): + """ref word idx -> utterance idx for sinr files""" + idx = [] + for u, t in enumerate(m["utts"]): idx += [u] * len(norm(t)) + return idx +BLOCK = 100 +def unit_of(m, ref_idx, uidx): + if m["kind"] == "sinr": return (m["split"], m["layout"], uidx[ref_idx]) + return (m["name"], ref_idx // BLOCK) +_cache = {} +def file_errors(det, name, vn): + """dict unit -> [errors, nref], plus counts S,D,I and events with times. cached on disk.""" + key = (det, name, vn) + f = f"{W}/cache/err/{det}/{vn}/{name}.pkl" + if os.path.exists(f): return pickle.load(open(f, "rb")) + m = MAN[name]; ref = norm(m["ref"]); segs = segs_for(det, name, vn); hy = hyp(det, name, segs); ev = align(ref, hy) + uidx = utt_index(m) if m["kind"] == "sinr" else None + units = {}; sdi = {"S": 0, "D": 0, "I": 0}; events = [] + last_ref = 0 + for kind, rj, hj in ev: + if rj is not None: last_ref = rj + if kind == "C": + u = unit_of(m, rj, uidx); units.setdefault(u, [0, 0])[1] += 1; continue + j = rj if rj is not None else last_ref + u = unit_of(m, min(j, len(ref) - 1), uidx); units.setdefault(u, [0, 0]) + units[u][0] += 1 + if rj is not None: units[u][1] += 1 + sdi[kind] += 1; events.append((kind, rj, hj)) + res = dict(units=units, sdi=sdi, events=events, nref=len(ref), nhyp=len(hy), nseg=len(segs)) + os.makedirs(os.path.dirname(f), exist_ok=True); pickle.dump(res, open(f, "wb")) + return res +def boot_diff(A, B, B_n=5000, seed=1): + """A, B: arrays [n_units, 2] (errors, nref) for the same units. Paired bootstrap of WER(B) - WER(A) in percentage points.""" + A = np.asarray(A, float); B = np.asarray(B, float); n = len(A) + rng = np.random.default_rng(seed); idx = rng.integers(0, n, size=(B_n, n)) + ea, na, eb, nb = A[idx, 0].sum(1), A[idx, 1].sum(1), B[idx, 0].sum(1), B[idx, 1].sum(1) + d = 100 * (eb / nb - ea / na) + pt = 100 * (B[:, 0].sum() / B[:, 1].sum() - A[:, 0].sum() / A[:, 1].sum()) + return pt, np.percentile(d, 2.5), np.percentile(d, 97.5) +def boot_ci(A, B_n=5000, seed=1): + A = np.asarray(A, float); n = len(A); rng = np.random.default_rng(seed); idx = rng.integers(0, n, size=(B_n, n)) + w = 100 * A[idx, 0].sum(1) / A[idx, 1].sum(1) + return 100 * A[:, 0].sum() / A[:, 1].sum(), np.percentile(w, 2.5), np.percentile(w, 97.5) +def group_units(det, vn, names): + """merge unit dicts of several files; returns (unit keys, array [n,2]) with consistent order, summed S/D/I""" + tot = {}; sdi = {"S": 0, "D": 0, "I": 0} + for nm in names: + r = file_errors(det, nm, vn) + for u, (e, n) in r["units"].items(): + t = tot.setdefault(u, [0, 0]); t[0] += e; t[1] += n + for k in sdi: sdi[k] += r["sdi"][k] + return tot, sdi +def paired(det, va, vb, names): + ta, sa = group_units(det, va, names); tb, sb = group_units(det, vb, names) + keys = sorted(set(ta) | set(tb), key=str) + A = [ta.get(k, [0, 0]) for k in keys]; B = [tb.get(k, [0, 0]) for k in keys] + return A, B, sa, sb diff --git a/scripts/vad_bench/trim_regression/scripts/mkcorpus.py b/scripts/vad_bench/trim_regression/scripts/mkcorpus.py new file mode 100644 index 0000000..1159e06 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/mkcorpus.py @@ -0,0 +1,69 @@ +#!/usr/bin/env python3 +"""Build the sets. usage: mkcorpus.py DATA + talks: tune = JamesCameron, JaneMcGonigal, DanBarber, RobertGupta; heldout = the others (>=30 s) + sn_*: LibriSpeech test-clean (every 13th utt; 40 speakers) split by SPEAKER (even index in sorted speaker list = tune, odd = heldout); + layouts of 6 utterances with 0.3-2.0 s gaps (the PR's construction); each layout in 3 noise seeds x + conditions clean, white{20,10,5,0}, pink{20,10,5,0} (SNR against the speech power). + heldout: 10 layouts (60 utts), tune: 4 layouts (24 utts). + ins_*: 90 s of a talk with 60 s of noise inserted at a quiet point (the PR's construction).""" +import glob, json, os, sys +import numpy as np, soundfile as sf +D = sys.argv[1]; SR = 16000 +sys.path.insert(0, os.path.dirname(__file__)) +from gens import GEN, pink, rms, at_db +TUNE_TALKS = {"JamesCameron-merged", "JaneMcGonigal-merged", "DanBarber-merged", "RobertGupta-merged"} +man = [] +for wav in sorted(glob.glob(f"{D}/talk_*.wav")): + name = os.path.basename(wav)[5:-4] + x, sr = sf.read(wav, dtype="float32") + if len(x) / SR < 60: continue + split = "tune" if name in TUNE_TALKS else "heldout" + man.append(dict(name=f"talk_{name}", kind="talk", split=split, ref=open(wav[:-4] + ".txt").read().strip(), dur=len(x) / SR)) + if len(x) < 150 * SR: continue + ex = x[60 * SR:150 * SR]; db = 20 * np.log10(rms(ex)) + fr = len(ex) // 1600; e = 10 * np.log10((ex[:fr * 1600].reshape(fr, 1600) ** 2).mean(1) + 1e-12) + c = [(e[i:i + 10].mean(), i) for i in range(250, 650)]; _, i0 = min(c); cut = (i0 + 5) * 1600 + for ti, (tn, rel) in enumerate([("white", -20), ("pink", -20), ("clicks", -20), ("music", -20), ("white", -35), ("music", -35), + ("white", -5), ("pink", -5), ("clicks", -5), ("music", -5)]): + r = np.random.default_rng(1000 + 17 * ti + len(name)) + n = np.clip(at_db(GEN[tn](60 * SR, r), db + rel), -1, 1) + y = np.concatenate([ex[:cut], n, ex[cut:]]).astype(np.float32) + nm = f"ins_{name}_{tn}{rel}" + sf.write(f"{D}/{nm}.wav", y, SR, subtype="PCM_16") + man.append(dict(name=nm, kind="insert", split=split, talk=name, noise=tn, level=rel, ins=[cut / SR, cut / SR + 60])) + +U = [u.astype(np.float32) / 32768 for u in np.load(f"{D}/libri.npy", allow_pickle=True)] +L = json.load(open(f"{D}/libri.json")); T = L["text"]; SPK = L["speaker"] +spk = sorted(set(SPK)); tune_spk = set(spk[0::2]) +def trim(u): + fr = len(u) // 160; r = np.sqrt((u[:fr * 160].reshape(fr, 160) ** 2).mean(1)) + idx = np.where(r > r.max() * 0.01)[0]; return u[idx[0] * 160:(idx[-1] + 1) * 160] +rng = np.random.default_rng(7) +pools = {} +for split in ("tune", "heldout"): + ids = [i for i in range(len(U)) if (SPK[i] in tune_spk) == (split == "tune")] + rng.shuffle(ids); pools[split] = ids +NLAY = {"tune": 5, "heldout": 12} +CONDS = [("clean", None, None)] + [(f"white{s}", "white", s) for s in (20, 10, 5, 0)] + [(f"pink{s}", "pink", s) for s in (20, 10, 5, 0)] +for split in ("tune", "heldout"): + for k in range(NLAY[split]): + ids = pools[split][6 * k:6 * k + 6]; r = np.random.default_rng(2000 + k + (500 if split == "tune" else 0)) + parts = [r.standard_normal(int(r.uniform(.5, 1.5) * SR)) * 1e-3]; spans = []; t = len(parts[0]) / SR + for i in ids: + u = trim(U[i]); spans.append((t, t + len(u) / SR)); parts.append(u); t += len(u) / SR + g = r.standard_normal(int(r.uniform(.3, 2.0) * SR)) * 1e-3; parts.append(g); t += len(g) / SR + y = np.concatenate(parts).astype(np.float32) + sp = np.concatenate([y[int(s * SR):int(e * SR)] for s, e in spans]); P = (sp ** 2).mean() + for cond, kind, snr in CONDS: + for seed in (range(3) if kind else range(1)): + z = y + if kind: + rr = np.random.default_rng(3000 + 10 * k + seed + (777 if split == "tune" else 0)) + n = rr.standard_normal(len(y)) if kind == "white" else pink(len(y), rr) + z = np.clip(y + n * np.sqrt(P / 10 ** (snr / 10)), -1, 1).astype(np.float32) + nm = f"sn_{split}_{cond}_L{k:02d}s{seed}"; sf.write(f"{D}/{nm}.wav", z, SR, subtype="PCM_16") + man.append(dict(name=nm, kind="sinr", split=split, cond=cond, layout=k, seed=seed, dur=len(y) / SR, + spans=spans, utts=[T[i] for i in ids], speakers=[SPK[i] for i in ids], ref=" ".join(T[i] for i in ids))) +json.dump(man, open(f"{D}/manifest.json", "w")) +from collections import Counter +print(len(man), Counter((m["kind"], m["split"]) for m in man)) diff --git a/scripts/vad_bench/trim_regression/scripts/noise_ins.py b/scripts/vad_bench/trim_regression/scripts/noise_ins.py new file mode 100644 index 0000000..c77e903 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/noise_ins.py @@ -0,0 +1,18 @@ +"""Noise block of 60 s inside speech: seconds of the block that the decoder gets, and words invented in it (decode cache needed for words).""" +import sys, collections; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from lib import * +vs = sys.argv[1].split(","); dets = sys.argv[2].split(",") if len(sys.argv) > 2 else list(DET) +words = len(sys.argv) > 3 and sys.argv[3] == "words" +ins = [m for m in MAN.values() if m["kind"] == "insert"] +def ov(segs, a, b): return sum(max(0.0, min(e, b) - max(s, a)) for s, e in segs) +print(f"{len(ins)} insert files (9 talks x white,pink,clicks,music @-20; white,music @-35; white,pink,clicks,music @-5)") +print(f"{'det':6s}{'variant':10s}{'sec/file':>9s}" + ("{'words':>7s}{'files>0':>8s}" if words else "")) +for det in dets: + for v in vs: + secs = []; wn = 0; fw = 0; bytype = collections.defaultdict(list) + for m in ins: + a, b = m["ins"]; segs = segs_for(det, m["name"], v); s = ov(segs, a, b); secs.append(s) + bytype[f"{m['noise']}{m['level']}"].append(s) + if words: + h = hyp(det, m["name"], segs); k = sum(a + 0.5 <= t[1] <= b - 0.5 for t in h); wn += k; fw += k > 0 + print(f"{det:6s}{v:10s}{np.mean(secs):9.1f}" + (f"{wn:7d}{fw:8d}" if words else ""), " " + " ".join(f"{k}:{np.mean(x):.0f}" for k, x in sorted(bytype.items()))) diff --git a/scripts/vad_bench/trim_regression/scripts/noise_words.py b/scripts/vad_bench/trim_regression/scripts/noise_words.py new file mode 100644 index 0000000..29b2794 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/noise_words.py @@ -0,0 +1,22 @@ +"""Invented words in the 60 s noise block, on the 20 insert files that were decoded (4 talks x white-20, pink-5, music-20, clicks-5, white-5). +Only segments overlapping the block were decoded; words inside [block+0.5 s, block end-0.5 s] count (as in the PR).""" +import sys, re; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from lib import * +vs = sys.argv[1].split(","); dets = sys.argv[2].split(",") +INS = re.compile(r"ins_(JamesCameron|JaneMcGonigal|DanielKahneman|MichaelSpecter)-merged_(white-20|pink-5|music-20|clicks-5|white-5)$") +ins = [m for m in MAN.values() if m["kind"] == "insert" and INS.match(m["name"])] +def ov(segs, a, b): return sum(max(0.0, min(e, b) - max(s, a)) for s, e in segs) +print(f"{len(ins)} files, noise block 60 s each") +print("| detector | variant | noise sec decoded / file | invented words | files with words | words per file with noise decoded |") +print("|---|---|---|---|---|---|") +for det in dets: + for v in vs: + secs = []; wn = 0; fw = 0 + for m in ins: + a, b = m["ins"]; segs = segs_for(det, m["name"], v); secs.append(ov(segs, a, b)); c = dec.load_cache(det, m["name"]) + k = 0 + for s, e in segs: + if e <= a - 0.5 or s >= b + 0.5: continue + k += sum(a + 0.5 <= w[0] <= b - 0.5 for w in c[dec.key(s, e)]["words"]) + wn += k; fw += k > 0 + print(f"| {det} | {v} | {np.mean(secs):.1f} | {wn} | {fw}/{len(ins)} | |") diff --git a/scripts/vad_bench/trim_regression/scripts/pooled.py b/scripts/vad_bench/trim_regression/scripts/pooled.py new file mode 100644 index 0000000..4543a53 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/pooled.py @@ -0,0 +1,18 @@ +"""Pooled paired comparison across the three detectors (units = (det, utterance) for sinr; (det, talk block) for talks).""" +import sys; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +def pool(dets, va, vb, kind, conds=None, split=None): + A = []; B = [] + for det in dets: + nm = names("talk", split) if kind == "talk" else names("sinr", split, conds) + nm = [n for n in nm if all(dec.key(a, b) in dec.load_cache(det, n) for v in (va, vb) for a, b in segs_for(det, n, v))] + a, b, _, _ = paired(det, va, vb, nm); A += a; B += b + return A, B +pairs = [("T0", "T0.3"), ("T0.3", "T0.5"), ("T0.3", "P0.5/0.3"), ("T0.3", "E0.5"), ("T0", "T0.5"), ("T0", "E0.5"), ("T0", "P0.5/0.3")] +N3 = ["white5", "pink5", "pink0"] +print("| comparison (B vs A) | set | words | WER A | WER B | delta (pp) | 95% CI |"); print("|---|---|---|---|---|---|---|") +for va, vb in pairs: + for label, kind, conds, split in (("noisy sinr 3 dets, tune+heldout", "sinr", N3, None), ("talks 3 dets, all 9", "talk", None, None)): + A, B = pool(["redux", "ultra", "v3"], va, vb, kind, conds, split) + pt, lo, hi = boot_diff(A, B); wa = boot_ci(A)[0]; wb = boot_ci(B)[0] + print(f"| {vb} vs {va} | {label} | {int(sum(x[1] for x in A))} | {wa:.2f} | {wb:.2f} | {pt:+.2f} | {lo:+.2f}..{hi:+.2f}{' *' if lo > 0 or hi < 0 else ''} |") diff --git a/scripts/vad_bench/trim_regression/scripts/probe.py b/scripts/vad_bench/trim_regression/scripts/probe.py new file mode 100644 index 0000000..a859796 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/probe.py @@ -0,0 +1,19 @@ +import os, subprocess, sys +from concurrent.futures import ThreadPoolExecutor +sys.path.insert(0, os.path.dirname(__file__)); from common import * +jobs = [] +for det, (model, sil) in DET.items(): + os.makedirs(f"{W}/probs/{det}", exist_ok=True) + for m in [x for x in manifest() if x.get("dur", 99) > 30]: + out = f"{W}/probs/{det}/{m['name']}.f32" + if os.path.exists(out): continue + cmd = [CLI, "transcribe", "--model", model, "--input", f"{W}/data/{m['name']}.wav", "--json", "--threads", "2", "--vad"] + (["--vad-model", sil] if sil else []) + jobs.append((out, cmd)) +def run(j): + out, cmd = j + env = dict(os.environ, PK_SEGMENTS="/dev/null", PK_PROBOUT=out + ".tmp") + r = subprocess.run(cmd, capture_output=True, text=True, env=env) + if r.returncode == 0 and os.path.exists(out + ".tmp"): os.rename(out + ".tmp", out) + else: print("FAIL", out, r.stderr[-200:], flush=True) +print(len(jobs), "jobs", flush=True) +with ThreadPoolExecutor(int(sys.argv[1]) if len(sys.argv) > 1 else 8) as ex: list(ex.map(run, jobs)) diff --git a/scripts/vad_bench/trim_regression/scripts/ready.py b/scripts/vad_bench/trim_regression/scripts/ready.py new file mode 100644 index 0000000..161aa22 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/ready.py @@ -0,0 +1,10 @@ +import sys, os; sys.path.insert(0, os.path.dirname(__file__)) +from groups import * +det, kind, vs = sys.argv[1], sys.argv[2], sys.argv[3].split(",") +ok = tot = 0 +for m in MAN.values(): + if m["kind"] != kind or m.get("dur", 99) <= 30: continue + if len(sys.argv) > 4 and m["kind"] == "sinr" and m["cond"] not in sys.argv[4].split(","): continue + tot += 1; c = dec.load_cache(det, m["name"]) + ok += all(dec.key(s, e) in c for v in vs for s, e in segs_for(det, m["name"], v)) +print(det, kind, ok, "/", tot) diff --git a/scripts/vad_bench/trim_regression/scripts/removed.py b/scripts/vad_bench/trim_regression/scripts/removed.py new file mode 100644 index 0000000..25b33fa --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/removed.py @@ -0,0 +1,15 @@ +import sys; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +for det in DET: + for kind, nm in (("talk", names("talk")), ("sinr-noisy", names("sinr", None, W_ + P_)), ("sinr-clean", names("sinr", None, ["clean"]))): + rem = []; nseg = 0; tot = 0.0 + for n in nm: + s0 = segs_for(det, n, "T0"); s3 = segs_for(det, n, "T0.3") + assert len(s0) == len(s3) + for (a, b), (c, d) in zip(s0, s3): + nseg += 1; tot += b - a + for x in (c - a, b - d): + rem.append(x) + rem = np.array(rem) + # exclude file start/end edges? keep all + print(f"{det:6s}{kind:11s} segments {nseg:5d} mean len {tot/nseg:5.1f}s | edges removed: >0 {100*(rem>1e-6).mean():4.1f}% >=0.1s {100*(rem>=0.1).mean():4.1f}% >=0.5s {100*(rem>=0.5).mean():4.1f}% >=1s {100*(rem>=1).mean():4.1f}% >=2s {100*(rem>=2).mean():4.1f}% mean {rem.mean():.2f}s p90 {np.percentile(rem,90):.2f}s") diff --git a/scripts/vad_bench/trim_regression/scripts/report_cand.py b/scripts/vad_bench/trim_regression/scripts/report_cand.py new file mode 100644 index 0000000..b321f10 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/report_cand.py @@ -0,0 +1,30 @@ +"""usage: report_cand.py DET SPLIT VARIANTS BASE columns: talks, clean, noisy (the SNR conditions listed), pink0, combined (talks+clean+noisy4).""" +import sys; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +det, split, vs, base = sys.argv[1], sys.argv[2], sys.argv[3].split(","), sys.argv[4] +s = None if split == "all" else split +N4 = sys.argv[5].split(",") if len(sys.argv) > 5 else ["white5", "white0", "pink5", "pink0"] +def have(nm): + out = [] + for n in nm: + c = dec.load_cache(det, n) + if all(dec.key(a, b) in c for v in [base] + vs for a, b in segs_for(det, n, v)): out.append(n) + return out +cols = {"talks": names("talk", s), "clean": names("sinr", s, ["clean"]), "noisy": names("sinr", s, N4), "pink0": names("sinr", s, ["pink0"]), "white0": names("sinr", s, ["white0"]) if "white0" in N4 else []} +cols = {k: have(v) for k, v in cols.items()} +cols["combined"] = cols["talks"] + cols["clean"] + cols["noisy"] +print(f"### {det}, {split}: WER % (delta vs {base}; paired bootstrap 95% CI; * = CI excludes 0)") +print("| variant | " + " | ".join(cols) + " |"); print("|---|" + "---|" * len(cols)) +nref = {} +for v in [base] + [x for x in vs if x != base]: + cells = [] + for c, nm in cols.items(): + if not nm: cells.append("-"); continue + try: + A, B, sa, sb = paired(det, base, v, nm); w = boot_ci(B)[0]; nref[c] = int(sum(x[1] for x in A)) + if v == base: cells.append(f"{w:.2f}") + else: + pt, lo, hi = boot_diff(A, B); cells.append(f"{w:.2f} ({pt:+.2f}; {lo:+.2f}..{hi:+.2f}){'*' if lo > 0 or hi < 0 else ''}") + except (KeyError, FileNotFoundError): cells.append("n/a") + print(f"| {v} | " + " | ".join(cells) + " |") +print("| ref words | " + " | ".join(str(nref.get(c, "")) for c in cols) + " |") diff --git a/scripts/vad_bench/trim_regression/scripts/report_grid.py b/scripts/vad_bench/trim_regression/scripts/report_grid.py new file mode 100644 index 0000000..e3e444a --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/report_grid.py @@ -0,0 +1,22 @@ +"""usage: report_grid.py DET SPLIT VARIANTS [BASE=T0] rows = variants; columns = groups. WER and paired delta vs BASE with 95% bootstrap CI (* = CI excludes 0).""" +import sys; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +det, split, vs = sys.argv[1], sys.argv[2], sys.argv[3].split(","); base = sys.argv[4] if len(sys.argv) > 4 else "T0" +cols = [f"talks[{split}]", f"clean[{split}]", f"white[{split}]", f"pink[{split}]", f"pink0[{split}]", f"white0[{split}]", f"noisy-all[{split}]", f"sinr-all[{split}]"] +gs = G(split) +print(f"### {det}, {split}: WER % (delta vs {base}, paired bootstrap 95% CI)") +print("| variant | " + " | ".join(c.replace(f"[{split}]", "") for c in cols) + " |") +print("|---|" + "---|" * len(cols)) +nrefs = {} +for v in [base] + [x for x in vs if x != base]: + cells = [] + for c in cols: + try: + A, B, sa, sb = paired(det, base, v, gs[c]); w = boot_ci(B)[0] + if v == base: cells.append(f"{w:.2f}") + else: + pt, lo, hi = boot_diff(A, B); cells.append(f"{w:.2f} ({pt:+.2f}; {lo:+.2f}..{hi:+.2f}){'*' if lo > 0 or hi < 0 else ''}") + nrefs[c] = int(sum(x[1] for x in A)) + except (KeyError, FileNotFoundError): cells.append("n/a") + print(f"| {v} | " + " | ".join(cells) + " |") +print("| words in reference | " + " | ".join(str(nrefs.get(c, "")) for c in cols) + " |") diff --git a/scripts/vad_bench/trim_regression/scripts/run.py b/scripts/vad_bench/trim_regression/scripts/run.py new file mode 100644 index 0000000..4b5ccb7 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/run.py @@ -0,0 +1,33 @@ +"""usage: run.py VARIANTS(comma) KINDS(comma: talk,sinr,insert) [workers] [threads] [--conds a,b] [--only regex]""" +import os, re, sys, argparse, soundfile as sf +from concurrent.futures import ThreadPoolExecutor +sys.path.insert(0, os.path.dirname(__file__)); from common import *; from seg import *; import dec, variants +ap = argparse.ArgumentParser(); ap.add_argument("variants"); ap.add_argument("kinds") +ap.add_argument("--workers", type=int, default=4); ap.add_argument("--threads", type=int, default=4) +ap.add_argument("--conds", default=""); ap.add_argument("--only", default=""); ap.add_argument("--dets", default="ultra,redux,v3") +ap.add_argument("--noise-only", action="store_true", help="inserts: decode only segments overlapping the noise block") +a = ap.parse_args() +vs = a.variants.split(","); kinds = a.kinds.split(","); conds = a.conds.split(",") if a.conds else None +files = [m for m in manifest() if m["kind"] in kinds and m.get("dur", 99) > 30 and (conds is None or m["kind"] != "sinr" or m["cond"] in conds) + and (not a.only or re.search(a.only, m["name"]))] +def need(det, m): + total = sf.info(f"{W}/data/{m['name']}.wav").frames / 16000 + p = probs(det, m["name"]); out = set() + for vn in vs: + for s, e in segments(p, total, OPTS[det], variants.get(vn, det)): + if a.noise_only and m["kind"] == "insert": + lo, hi = m["ins"] + if e <= lo - 0.5 or s >= hi + 0.5: continue + out.add((round(s, 4), round(e, 4))) + return sorted(out) +tasks = [(det, m) for det in a.dets.split(",") for m in files] +def go(t): + det, m = t + try: dec.run_segments(det, m["name"], need(det, m), threads=a.threads) + except Exception as ex: print("FAIL", det, m["name"], ex, flush=True) + return 1 +print(len(tasks), "tasks", flush=True) +with ThreadPoolExecutor(a.workers) as ex: + for i, _ in enumerate(ex.map(go, tasks)): + if i % 50 == 0: print(i, flush=True) +print("done", flush=True) diff --git a/scripts/vad_bench/trim_regression/scripts/seg.py b/scripts/vad_bench/trim_regression/scripts/seg.py new file mode 100644 index 0000000..4fd2dac --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/seg.py @@ -0,0 +1,73 @@ +"""Python replica of pk::segment_by_vad (master 2de154c), generalised with candidate trim rules. +Variant dict keys: pre, post (seconds kept before first / after last speech frame; None = no trim), +minlen (after trimming extend symmetrically, within the cut, to at least this long), +min_edge (trim an edge only if the audio removed there is at least this long).""" +import math +import numpy as np + +def mask(p, n, o): + fs = o["frame_sec"] + sp = np.zeros(n, dtype=bool); k = min(n, len(p)); sp[:k] = p[:k] >= o["threshold"] + def runs(val): + out = []; f = 0 + while f < n: + if sp[f] != val: f += 1; continue + e = f + while e < n and sp[e] == val: e += 1 + out.append((f, e)); f = e + return out + for a, b in runs(False): + if a > 0 and b < n and (b - a) * fs + 1e-9 < o["bridge_sec"]: sp[a:b] = True + for a, b in runs(True): + if (b - a) * fs + 1e-9 < o["min_speech_sec"]: sp[a:b] = False + return sp, runs + +def segments(p, total, o, v=None): + """v: None or {'pre':..,'post':..,'minlen':..,'min_edge':..}; pre=post=0 with v given means trim to speech exactly.""" + fs = o["frame_sec"] + if total <= o["max_seg_sec"]: return [(0.0, total)] + n = max(0, int(math.ceil(total / fs - 1e-9))) + max_f = max(2, int(math.floor(o["max_seg_sec"] / fs + 1e-9))) + min_f = min(max_f - 1, max(1, int(math.ceil(o["min_seg_sec"] / fs - 1e-9)))) + pause_f = max(1, int(math.ceil(o["min_pause_sec"] / fs - 1e-9))) + sp, runs = mask(p, n, o) + pauses = [(a, b) for a, b in runs(False) if b - a >= pause_f] + cum = np.concatenate([[0], np.cumsum(sp)]) + out = [] + def emit(a, b, end_sec): + be = min(b, n) + if not (be > a and cum[be] - cum[a] > 0): return + s0, e0 = a * fs, end_sec + if v is not None and v.get("pre") is not None: + fa, fb = a, be + while fa < fb and not sp[fa]: fa += 1 + while fb > fa and not sp[fb - 1]: fb -= 1 + ts, te = fa * fs - v["pre"], fb * fs + v["post"] + me = v.get("min_edge", 0.0) or 0.0 + ns, ne = s0, e0 + if ts - s0 >= me and ts > s0: ns = ts + if e0 - te >= me and te < e0: ne = te + ml = v.get("minlen", 0.0) or 0.0 + if ne - ns < ml: + need = ml - (ne - ns); ns2 = max(s0, ns - need / 2); ne2 = min(e0, ne + need / 2) + # give the unused share to the other side + if ns2 - (ns - need / 2) < -1e-12: ne2 = min(e0, ne2 + (ns - need / 2 - ns2)) + if (ne + need / 2) - ne2 > 1e-12: ns2 = max(s0, ns2 - ((ne + need / 2) - ne2)) + ns, ne = ns2, ne2 + s0, e0 = ns, ne + out.append((s0, e0)) + s = 0 + while total - s * fs > o["max_seg_sec"] + 1e-9: + lo, hi = s + min_f, s + max_f + c = -1; c_mid = -1 + for a, b in pauses: + mid = (a + b) // 2 + if a >= lo and b <= hi: c = mid + if lo <= mid <= hi: c_mid = mid + if c < 0: c = c_mid + if c <= s: c = hi + emit(s, c, c * fs); s = c + emit(s, n, total) + return out + +def base(trim): return None if trim is None else {"pre": trim, "post": trim} diff --git a/scripts/vad_bench/trim_regression/scripts/segchange.py b/scripts/vad_bench/trim_regression/scripts/segchange.py new file mode 100644 index 0000000..910e252 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/segchange.py @@ -0,0 +1,27 @@ +"""Per segment: does the text change between two variants, and does the error count go up or down? (segments are 1:1 between variants.) +Errors are attributed to the segment of the hypothesis token (S, I) or, for D, of the preceding hypothesis token (first segment if none). +usage: segchange.py DET VA VB [kinds]""" +import sys, collections; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +from scipy.stats import binomtest +def seg_errors(det, name, vn): + segs = segs_for(det, name, vn); hy = hyp(det, name, segs); ev = align(norm(MAN[name]["ref"]), hy) + E = np.zeros(len(segs), int); last = 0; texts = [[] for _ in segs] + for t in hy: texts[t[3]].append(t[0]) + for kind, rj, hj in ev: + if hj is not None: last = hy[hj][3] + if kind in "SDI": E[last if kind == "D" else hy[hj][3]] += 1 + return E, [" ".join(t) for t in texts] +det, va, vb = sys.argv[1:4]; kinds = sys.argv[4].split(",") if len(sys.argv) > 4 else ["talk", "sinr"] +for kind in kinds: + nms = names(kind) if kind == "talk" else names("sinr", None, W_ + P_) + n = ch = up = down = 0; dsum = 0 + for nm in nms: + try: Ea, Ta = seg_errors(det, nm, va); Eb, Tb = seg_errors(det, nm, vb) + except KeyError: continue + for i in range(len(Ea)): + n += 1 + if Ta[i] != Tb[i]: + ch += 1; d = Eb[i] - Ea[i]; dsum += d; up += d > 0; down += d < 0 + p = binomtest(up, up + down, 0.5).pvalue if up + down else 1.0 + print(f"{det:6s}{kind:5s} {va}->{vb}: segments {n}, text changed {ch} ({100*ch/n:.1f}%); of those errors up {up}, down {down}, equal {ch-up-down}; net {dsum:+d}; sign test p={p:.2f}") diff --git a/scripts/vad_bench/trim_regression/scripts/table_trim.py b/scripts/vad_bench/trim_regression/scripts/table_trim.py new file mode 100644 index 0000000..53c0694 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/table_trim.py @@ -0,0 +1,22 @@ +import sys; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +vs = sys.argv[1].split(","); split = sys.argv[2]; dets = sys.argv[3].split(",") if len(sys.argv) > 3 else list(DET) +base = vs[0] +for det in dets: + print(f"\n#### {det} split={split} (WER %, delta vs {base} with paired bootstrap 95% CI; [S/D/I] counts)") + for g, nm in G(split).items(): + try: + row = [] + nref = None + for v in vs: + A, B, sa, sb = paired(det, base, v, nm) + w = boot_ci(B)[0] + if v == base: row.append(f"{w:5.2f} [{sa['S']}/{sa['D']}/{sa['I']}]") + else: + pt, lo, hi = boot_diff(A, B) + sig = "*" if lo > 0 or hi < 0 else " " + row.append(f"{w:5.2f} {pt:+5.2f}({lo:+5.2f},{hi:+5.2f}){sig} [{sb['S']}/{sb['D']}/{sb['I']}]") + nref = int(np.sum([x[1] for x in A])) + print(f"{g:20s} n_ref={nref:6d} units={len(A):4d} | " + " | ".join(row)) + except Exception as ex: + print(f"{g:20s} missing: {type(ex).__name__} {str(ex)[:60]}") diff --git a/scripts/vad_bench/trim_regression/scripts/vad_cover.py b/scripts/vad_bench/trim_regression/scripts/vad_cover.py new file mode 100644 index 0000000..d46a0ac --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/vad_cover.py @@ -0,0 +1,30 @@ +"""How far outside the VAD speech mask are the (correctly) decoded words of the untrimmed run? +For each correct word of the T0 decode: gap = distance from the word interval to the nearest speech frame of the smoothed mask +(0 = overlaps speech; >0 = the word lies outside the mask, before ('lead') or after ('lag') the speech run).""" +import sys, collections; sys.path.insert(0, __import__('os').path.dirname(__file__)) +from groups import * +dets = sys.argv[1].split(",") if len(sys.argv) > 1 else list(DET) +for det in dets: + o = OPTS[det]; fs = o["frame_sec"] + for kind, nm in (("talks", names("talk")), ("sinr clean", names("sinr", None, ["clean"])), ("sinr noisy", names("sinr", None, W_ + P_)), ("sinr pink0+white0", names("sinr", None, ["pink0", "white0"]))): + lead = []; lag = []; tot = 0 + for n in nm: + try: segs = segs_for(det, n, "T0"); hy = hyp(det, n, segs); ev = align(norm(MAN[n]["ref"]), hy) + except KeyError: continue + st = {hj: k for k, rj, hj in ev if hj is not None} + tt = total_sec(n); nf = int(np.ceil(tt / fs - 1e-9)); sp, _ = mask(probs(det, n), nf, o) + idx = np.where(sp)[0]; starts = idx * fs; ends = (idx + 1) * fs + for hj, (tok, s, e, si, cf) in enumerate(hy): + if st[hj] != "C": continue + tot += 1 + # nearest speech frame + k = np.searchsorted(starts, s) + # speech overlapping the word? + ov = np.any((starts < e) & (ends > s)) if len(idx) else False + if ov: lead.append(0.0); lag.append(0.0); continue + prev_end = ends[ends <= s].max() if np.any(ends <= s) else -9 + next_start = starts[starts >= e].min() if np.any(starts >= e) else 1e9 + lead.append(max(0.0, next_start - e) if next_start - e < s - prev_end else 0.0) # word is before the next speech run + lag.append(max(0.0, s - prev_end) if s - prev_end <= next_start - e else 0.0) # word is after the previous speech run + lead = np.array(lead); lag = np.array(lag) + print(f"{det:6s}{kind:18s} correct words {tot:6d} | outside the mask (any) {100*np.mean((lead>0)|(lag>0)):4.2f}% start-side>0.1s {100*np.mean(lead>0.1):4.2f}% >0.3s {100*np.mean(lead>0.3):4.2f}% end-side>0.1s {100*np.mean(lag>0.1):4.2f}% >0.3s {100*np.mean(lag>0.3):4.2f}%") diff --git a/scripts/vad_bench/trim_regression/scripts/vad_edges.py b/scripts/vad_bench/trim_regression/scripts/vad_edges.py new file mode 100644 index 0000000..0473c75 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/vad_edges.py @@ -0,0 +1,50 @@ +"""Onset/offset timing of the VAD against the true utterance spans (sinr files). No ASR.""" +import sys, os, numpy as np, collections +sys.path.insert(0, os.path.dirname(__file__)); from lib import * +files = [m for m in MAN.values() if m["kind"] == "sinr" and m["dur"] > 30] +def cond_family(c): return "clean" if c == "clean" else c +rows = collections.defaultdict(list) # (det, family, kind) -> list of values +prof = collections.defaultdict(lambda: collections.defaultdict(list)) # (det, fam, 'on'|'off') -> rel offset -> list of speech flags +REL = np.round(np.arange(-0.48, 0.97, 0.08), 2) +for det in DET: + o = OPTS[det]; fs = o["frame_sec"] + for m in files: + p = probs(det, m["name"]); total = m["dur"]; n = int(np.ceil(total / fs - 1e-9)) + sp, _ = mask(p, n, o); raw = np.zeros(n, bool); raw[:min(n, len(p))] = p[:n] >= o["threshold"] + segs = segments(p, total, o, None); cuts = [s for s, e in segs[1:]] + [e for s, e in segs[:-1]] + for ui, (us, ue) in enumerate(m["spans"]): + f0 = max(0, int((us - 0.5) / fs)); f1 = min(n, int(np.ceil(ue / fs))) + fr = [f for f in range(f0, f1) if sp[f] and (f + 1) * fs > us - 0.5] + first = next((f for f in range(f0, f1) if sp[f]), None) + g0 = max(0, int(us / fs)); g1 = min(n, int(np.ceil((ue + 0.5) / fs))) + last = next((f for f in range(g1 - 1, g0 - 1, -1) if sp[f]), None) + # edge utterance: next to a cut (cut within 1.5 s before start / after end) + edge_on = any(us - 3.0 <= c <= us for c in cuts); edge_off = any(ue <= c <= ue + 3.0 for c in cuts) + fam = cond_family(m["cond"]) + for tag, ok in (("all", True), ("edge", None)): + pass + late = None if first is None else first * fs - us + early = None if last is None else ue - (last + 1) * fs + for fam_ in (fam, "ALL-noisy" if fam != "clean" else "clean", "ALL"): + rows[(det, fam_, "on_all")].append(late); rows[(det, fam_, "off_all")].append(early) + if edge_on: rows[(det, fam_, "on_edge")].append(late) + if edge_off: rows[(det, fam_, "off_edge")].append(early) + for r in REL: + fo = int(np.floor((us + r) / fs)); fe = int(np.floor((ue + r - 0.0) / fs)) + if 0 <= fo < n: prof[(det, "ALL-noisy" if fam != "clean" else "clean", "on")][r].append(raw[fo]) + if 0 <= fe < n: prof[(det, "ALL-noisy" if fam != "clean" else "clean", "off")][r].append(raw[fe]) +def stats(v): + miss = sum(x is None for x in v); x = np.array([a for a in v if a is not None]) + return f"n={len(v):4d} miss={100*miss/len(v):4.1f}% median={np.median(x)*1000:5.0f}ms p90={np.percentile(x,90)*1000:5.0f} p95={np.percentile(x,95)*1000:5.0f} >0.3s={100*(x>0.3).mean():4.1f}% >0.5s={100*(x>0.5).mean():4.1f}%" +print("Lateness of the speech start (VAD smoothed mask start minus true utterance start) and earliness of the end (true end minus VAD end). Positive = the VAD cuts into the word.") +for fam in ("clean", "white20", "white10", "white5", "white0", "pink20", "pink10", "pink5", "pink0", "ALL-noisy"): + print(f"-- {fam}") + for det in DET: + for k in ("on_all", "on_edge", "off_all", "off_edge"): + if (det, fam, k) in rows: print(f" {det:6s}{k:8s}", stats(rows[(det, fam, k)])) +print("\nFraction of true onsets (offsets) at which the RAW frame (p>=0.5) at time t relative to the true onset (end) is speech. t in s.") +for fam in ("clean", "ALL-noisy"): + for kind in ("on", "off"): + print(f"-- {fam} {kind}set t=" + " ".join(f"{r:+.2f}" for r in REL[::2])) + for det in DET: + print(f" {det:6s} " + " ".join(f"{100*np.mean(prof[(det, fam, kind)][r]):5.0f}" for r in REL[::2])) diff --git a/scripts/vad_bench/trim_regression/scripts/validate.py b/scripts/vad_bench/trim_regression/scripts/validate.py new file mode 100644 index 0000000..acb1256 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/validate.py @@ -0,0 +1,26 @@ +import os, subprocess, sys, json, tempfile +sys.path.insert(0, os.path.dirname(__file__)); from common import *; from seg import *; import dec +tot = bad = wbad = 0 +for det in DET: + model, sil = DET[det] + for name in ["talk_GaryFlake-merged", "sn_heldout_pink0_L00s0", "sn_heldout_white5_L03s1", "sn_tune_clean_L02s0", "ins_GaryFlake-merged_pink-5"]: + wav = f"{W}/data/{name}.wav" + import soundfile as sf + total = sf.info(wav).frames / 16000 + p = probs(det, name) + for trim in (0.3, 0.0): + with tempfile.TemporaryDirectory(dir=f"{W}/tmp") as td: + so = f"{td}/o.txt" + cmd = [CLI, "transcribe", "--model", model, "--input", wav, "--json", "--threads", "6", "--vad", "--vad-trim", str(trim)] + (["--vad-model", sil] if sil else []) + r = subprocess.run(cmd, capture_output=True, text=True, env=dict(os.environ, PK_SEGOUT=so)) + ref = [l.split("\t") for l in open(so).read().split("\n") if l] if os.path.exists(so) else [] + clitext = json.loads(r.stdout)["text"] + mine = segments(p, total, OPTS[det], base(trim if trim > 0 else None)) + ok = len(ref) == len(mine) and all(abs(float(a[0]) - s) < 2e-4 and abs(float(a[1]) - e) < 2e-4 for a, (s, e) in zip(ref, mine)) + # text from the harness + cache = dec.run_segments(det, name, mine, threads=6) + txt = " ".join(" ".join(w[3] for w in cache[dec.key(s, e)]["words"]) for s, e in mine) + same = txt.split() == clitext.split() + tot += 1; bad += not ok; wbad += not same + print(det, name, trim, "segs", len(ref), len(mine), "bounds_match", ok, "text_match", same, flush=True) +print("segment mismatches", bad, "text mismatches", wbad, "of", tot) diff --git a/scripts/vad_bench/trim_regression/scripts/variants.py b/scripts/vad_bench/trim_regression/scripts/variants.py new file mode 100644 index 0000000..38b9cd3 --- /dev/null +++ b/scripts/vad_bench/trim_regression/scripts/variants.py @@ -0,0 +1,14 @@ +"""Variant registry: name -> dict (det -> v) or a plain v for all dets. v = None means the old cuts (--vad-trim 0).""" +def sym(x): return {"pre": x, "post": x} +V = {"T0": None} +for t in (0.1, 0.2, 0.25, 0.3, 0.35, 0.5, 1.0): V[f"T{t}"] = sym(t) +def get(name, det): + v = V[name] + return v[det] if isinstance(v, dict) and det in v else v +# null controls (grid-aligned and unaligned perturbations of the default), asymmetric pads, min removed length, min length after trim +for t in (0.32, 0.4): V[f"T{t}"] = sym(t) +for a, b in ((0.5, 0.3), (0.6, 0.3), (0.4, 0.2), (0.3, 0.5), (0.5, 0.2)): V[f"P{a}/{b}"] = {"pre": a, "post": b} +for x in (0.5, 1.0, 2.0): V[f"E{x}"] = {"pre": 0.3, "post": 0.3, "min_edge": x} +for x in (0.5, 1.0): V[f"E{x}p0.5"] = {"pre": 0.5, "post": 0.5, "min_edge": x} +for x in (3.0, 6.0): V[f"L{x:g}"] = {"pre": 0.3, "post": 0.3, "minlen": x} +for a, b in ((0.5, 0.4), (0.4, 0.3), (0.4, 0.4), (0.6, 0.4)): V[f"P{a}/{b}"] = {"pre": a, "post": b} diff --git a/scripts/vad_bench/trim_regression/throwaway_cli_patch.diff b/scripts/vad_bench/trim_regression/throwaway_cli_patch.diff new file mode 100644 index 0000000..d60b008 --- /dev/null +++ b/scripts/vad_bench/trim_regression/throwaway_cli_patch.diff @@ -0,0 +1,50 @@ +# MEASUREMENT ONLY. This patch is not part of the product and must not be merged. +# It adds three environment variables to a build of parakeet-cli so that a script can +# (PK_PROBOUT) dump the VAD probabilities, (PK_SEGMENTS) replace the segmenter by a list of +# "start end" lines, and (PK_SEGOUT) print one line per decoded slice with its words. +# Apply with: git apply throwaway_cli_patch.diff (against master at 2de154c, src/model.cpp) +diff --git a/src/model.cpp b/src/model.cpp +index d1d25a5..2044ffc 100644 +--- a/src/model.cpp ++++ b/src/model.cpp +@@ -1,3 +1,5 @@ ++#include ++#include + #include "model.hpp" + + #include "audio_io.hpp" +@@ -435,8 +437,19 @@ std::vector vad_slices(const Model& m, const std::vector& pcm16k, + if (!(enc_frame_sec > 0.0) || !std::isfinite(enc_frame_sec)) + throw std::runtime_error("invalid encoder frame size"); + const double total_sec = (double)pcm16k.size() / 16000.0; +- const std::vector p = ext ? (*ext)(pcm16k) : m.vad_probabilities(pcm16k); +- const std::vector segs = segment_by_vad(p, total_sec, opts); ++ std::vector segs; ++ const char* sf = std::getenv("PK_SEGMENTS"); // THROWAWAY: segments from a text file "start end" per line ++ const char* po = std::getenv("PK_PROBOUT"); // THROWAWAY: dump the probabilities (float32) ++ std::vector p; ++ if (!sf || po) p = ext ? (*ext)(pcm16k) : m.vad_probabilities(pcm16k); ++ if (po) { FILE* f = std::fopen(po, "wb"); std::fwrite(p.data(), sizeof(float), p.size(), f); std::fclose(f); } ++ if (sf) { ++ FILE* f = std::fopen(sf, "r"); double a, b; ++ while (f && std::fscanf(f, "%lf %lf", &a, &b) == 2) segs.push_back({a, b}); ++ if (f) std::fclose(f); ++ } else { ++ segs = segment_by_vad(p, total_sec, opts); ++ } + std::vector out; + const size_t n = pcm16k.size(); + auto at = [&](double sec) { +@@ -512,6 +525,12 @@ Transcription Model::transcribe_pcm_vad_with_timestamps(const std::vector + Transcription& t = parts[i]; + if (filter.active()) all.dropped_words += apply_word_filter(t, filter); + for (Word& w : t.words) { w.start += (float)s.start_sec; w.end += (float)s.start_sec; } ++ if (const char* so = std::getenv("PK_SEGOUT")) { // THROWAWAY: one line per slice ++ FILE* f = std::fopen(so, i == 0 ? "w" : "a"); ++ std::fprintf(f, "%.6f\t%.6f\t", s.start_sec, s.start_sec + (double)s.pcm.size() / 16000.0); ++ for (size_t k = 0; k < t.words.size(); ++k) std::fprintf(f, "%s%.3f|%.3f|%.3f|%s", k ? " " : "", t.words[k].start, t.words[k].end, t.words[k].conf, t.words[k].text.c_str()); ++ std::fprintf(f, "\n"); std::fclose(f); ++ } + for (TokenInfo& k : t.tokens) k.frame += s.start_frame; + if (!t.text.empty()) { + if (!all.text.empty()) all.text += ' ';