From c117f1cd6ed863784832892b5ce4e61d6bcfd740 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Tue, 6 Oct 2026 07:39:51 +0000 Subject: [PATCH 1/3] feat: add an opt-in run gate to the VAD segmenter The Ultra and Redux heads call steady noise speech, but a noise run sits on a plateau while a speech run sits high. Add SegmenterOpts::run_gate (default 0 = off): a speech run, a stretch of frames with p >= threshold found before bridging, is dropped when the median of its frame probabilities is below the gate. A median equal to the gate keeps the run. The gate acts first, so dropped frames are silence for bridging, min_speech, pauses and trim. It works for any detector and frame size, in both modes. The streaming event tracker has no run median and ignores it. Expose it as the "run_gate" key (a number in [0, 1)) of the VAD options JSON, so the vad functions and the transcribe functions that take VAD options accept it, and as --run-gate on `parakeet-cli vad` and --vad-run-gate on `transcribe --vad`. A VAD stream refuses a non-zero value instead of ignoring it. With the gate off the output is unchanged byte for byte. Tests use synthetic probability streams at 0.08 s and 0.032 s frames with exact expected runs, the boundary, bridged gaps, a low median under a high peak, the trim, the option parser and CLI errors. A model test runs the gate on a clip with a seeded noise stretch for the head slice, Ultra, Redux and Silero. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] --- examples/cli/main.cpp | 13 +- include/parakeet_capi.h | 10 +- src/parakeet_capi.cpp | 1 + src/vad_json.cpp | 5 +- src/vad_json.hpp | 4 +- src/vad_segmenter.cpp | 28 ++- src/vad_segmenter.hpp | 18 +- tests/CMakeLists.txt | 13 ++ tests/test_capi_vad_silero.cpp | 8 + tests/test_vad_options.cpp | 24 +++ tests/test_vad_run_gate.cpp | 313 ++++++++++++++++++++++++++++++ tests/test_vad_run_gate_model.cpp | 271 ++++++++++++++++++++++++++ 12 files changed, 696 insertions(+), 12 deletions(-) create mode 100644 tests/test_vad_run_gate.cpp create mode 100644 tests/test_vad_run_gate_model.cpp diff --git a/examples/cli/main.cpp b/examples/cli/main.cpp index c60a46a..6e32deb 100644 --- a/examples/cli/main.cpp +++ b/examples/cli/main.cpp @@ -373,9 +373,10 @@ static int cmd_transcribe_stream(const std::string& model, const std::string& in // Segmenter options the user set on the command line. A value left unset keeps // the default of the VAD in use (Ultra/Redux head or Silero). struct VadOverrides { - std::optional threshold, min_pause, min_speech, max_seg, pad, trim; + std::optional threshold, min_pause, min_speech, max_seg, pad, trim, run_gate; void apply(pk::SegmenterOpts& o) const { if (trim) o.trim_sec = *trim; + if (run_gate) o.run_gate = (float)*run_gate; if (threshold) o.threshold = (float)*threshold; if (min_pause) o.min_pause_sec = *min_pause; if (min_speech) o.min_speech_sec = *min_speech; @@ -525,6 +526,9 @@ static int cmd_transcribe(int argc, char** argv) { } else if (std::strcmp(argv[i], "--vad-trim") == 0 && i + 1 < argc) { if (!parse_nonneg(argv[++i], d)) { std::fprintf(stderr, "parakeet-cli: --vad-trim must be >= 0 (0 = keep the whole cuts)\n"); return 2; } vad_ov.trim = d; + } else if (std::strcmp(argv[i], "--vad-run-gate") == 0 && i + 1 < argc) { + if (!parse_nonneg(argv[++i], d) || d >= 1.0) { std::fprintf(stderr, "parakeet-cli: --vad-run-gate must be in [0,1) (0 = off)\n"); return 2; } + vad_ov.run_gate = d; } else if (std::strcmp(argv[i], "--min-local-conf") == 0 && i + 1 < argc) { if (!parse_nonneg(argv[++i], d) || d > 1.0) { std::fprintf(stderr, "parakeet-cli: --min-local-conf must be in [0,1] (0 = off)\n"); return 2; } word_filter.min_local_conf = (float)d; @@ -545,7 +549,7 @@ static int cmd_transcribe(int argc, char** argv) { "[--threads N] [--json] " "[--component NAME] " "[--vad [--vad-model ] [--vad-component NAME] [--vad-threshold F=0.5] [--vad-min-pause SEC] " - "[--vad-min-speech SEC] [--vad-max-seg SEC=30] [--vad-trim SEC=0.3]] " + "[--vad-min-speech SEC] [--vad-max-seg SEC=30] [--vad-trim SEC=0.3] [--vad-run-gate P=0]] " "[--min-local-conf F [--local-radius SEC=5]] [--drop-punct-only] " "[--beam-size N [--nbest N] [--no-score-norm]]\n"); return 2; @@ -2303,6 +2307,9 @@ static int cmd_vad(int argc, char** argv) { } else if (std::strcmp(argv[i], "--trim") == 0 && i + 1 < argc) { if (!num(argv[++i], d, true)) return bad("--trim must be >= 0"); ov.trim = d; + } else if (std::strcmp(argv[i], "--run-gate") == 0 && i + 1 < argc) { + if (!num(argv[++i], d, true) || d >= 1.0) return bad("--run-gate must be in [0,1) (0 = off)"); + ov.run_gate = d; } else if (std::strcmp(argv[i], "--mode") == 0 && i + 1 < argc) { const char* v = argv[++i]; if (std::strcmp(v, "speech") == 0) mode = pk::VadRequest::Mode::kSpeech; @@ -2316,7 +2323,7 @@ static int cmd_vad(int argc, char** argv) { std::fprintf(stderr, "usage: parakeet-cli vad --model --input " "[--component NAME] [--threshold F=0.5] [--min-pause SEC] [--min-speech SEC] [--speech-pad SEC] " - "[--max-segment SEC=30] [--trim SEC=0.3] [--mode speech|segments] [--probabilities] [--threads N]\n"); + "[--max-segment SEC=30] [--trim SEC=0.3] [--run-gate P=0] [--mode speech|segments] [--probabilities] [--threads N]\n"); return 2; } if (threads > 0) pk::set_num_threads(threads); diff --git a/include/parakeet_capi.h b/include/parakeet_capi.h index e38db5e..4539b7a 100644 --- a/include/parakeet_capi.h +++ b/include/parakeet_capi.h @@ -269,6 +269,12 @@ char* parakeet_capi_transcribe_path_json_vad(parakeet_ctx* ctx, const char* wav_ // "trim" seconds >= 0; "segments" mode: each segment shrinks to its // first and last speech frame plus this much; default 0.3; // 0 keeps the whole cuts +// "run_gate" probability in [0, 1); both modes: a speech run (frames +// with p >= threshold, before bridging) whose median frame +// probability is below this is dropped; a median equal to it +// keeps the run; default 0 = off. Meant for the Ultra and +// Redux heads (docs/vad.md). Offline only: a VAD stream +// (parakeet_capi_vad_stream_begin) refuses a non-zero value // "mode" "speech" (default) or "segments" // "probabilities" true to add the per-frame probabilities; default false // Unknown keys and out-of-range values are errors. The Silero defaults are the @@ -309,7 +315,7 @@ char* parakeet_capi_vad_path_json(parakeet_ctx* ctx, const char* wav_path, // Silero context, or NULL to use the ASR model's own head (then the result is // as parakeet_capi_transcribe_path_json_vad, with the options below). Options // are the JSON object of parakeet_capi_vad_pcm_json; only "threshold", -// "min_pause", "min_speech", "max_segment" and "trim" are used here (the other +// "min_pause", "min_speech", "max_segment", "trim" and "run_gate" are used here (the other // keys of that object are accepted and ignored), plus the word filter keys of // parakeet_capi_transcribe_path_json_with, which apply to each segment on its // own (and to the whole file when it is at most max_segment seconds). NULL or @@ -329,7 +335,7 @@ typedef struct parakeet_vad_stream parakeet_vad_stream; // `vad` is a Silero context; `sample_rate` is 16000 or 8000 (no resampling in a // stream). `options_json` as in parakeet_capi_vad_pcm_json, but "mode" must be -// "speech" (the default). NULL on error (see vad's last error). +// "speech" (the default) and "run_gate" must be 0 (a stream has no run median). NULL on error (see vad's last error). parakeet_vad_stream* parakeet_capi_vad_stream_begin(parakeet_ctx* vad, int sample_rate, const char* options_json); diff --git a/src/parakeet_capi.cpp b/src/parakeet_capi.cpp index 88ad0e2..30a9454 100644 --- a/src/parakeet_capi.cpp +++ b/src/parakeet_capi.cpp @@ -870,6 +870,7 @@ extern "C" parakeet_vad_stream* parakeet_capi_vad_stream_begin(parakeet_ctx* vad std::string err; if (!pk::parse_vad_options(options_json, req, err, pk::VadKind::kSilero)) { vad->last_error = err; return nullptr; } if (req.mode != pk::VadRequest::Mode::kSpeech) { vad->last_error = "VAD streams support mode \"speech\" only"; return nullptr; } + if (req.opts.run_gate > 0.0f) { vad->last_error = "VAD option run_gate is offline only: a stream has no run median"; return nullptr; } req.opts.frame_sec = pk::kSileroFrameSec; auto* s = new (std::nothrow) parakeet_vad_stream(); if (!s) { vad->last_error = "out of memory"; return nullptr; } diff --git a/src/vad_json.cpp b/src/vad_json.cpp index ecf5650..4fa3cb3 100644 --- a/src/vad_json.cpp +++ b/src/vad_json.cpp @@ -73,7 +73,7 @@ bool parse_options(const char* json, VadRequest& req, std::string& err, bool vad c.ws(); const std::string opt = std::string(what) + " " + key; const bool is_seg_num = key == "threshold" || key == "min_pause" || key == "min_speech" || - key == "max_segment" || key == "speech_pad" || key == "trim"; + key == "max_segment" || key == "speech_pad" || key == "trim" || key == "run_gate"; if (vad_keys && key == "mode") { std::string v; if (!parse_string(c, v)) { err = "invalid " + opt + ": expected a string"; return false; } @@ -93,6 +93,9 @@ bool parse_options(const char* json, VadRequest& req, std::string& err, bool vad if (key == "threshold") { if (!(v > 0.0 && v <= 1.0)) { err = "invalid " + opt + ": must be in (0, 1]"; return false; } req.opts.threshold = (float)v; + } else if (key == "run_gate") { + if (!(v >= 0.0 && v < 1.0)) { err = "invalid " + opt + ": must be in [0, 1) (0 = off)"; return false; } + req.opts.run_gate = (float)v; } else if (key == "min_local_conf") { if (!(v >= 0.0 && v <= 1.0)) { err = "invalid " + opt + ": must be in [0, 1] (0 = off)"; return false; } req.filter.min_local_conf = (float)v; diff --git a/src/vad_json.hpp b/src/vad_json.hpp index 5d62423..7f70482 100644 --- a/src/vad_json.hpp +++ b/src/vad_json.hpp @@ -25,7 +25,9 @@ struct VadRequest { // Parses an options document: a flat JSON object, or NULL / "" for the // defaults. Keys: "threshold" (0 < x <= 1), "min_pause", "min_speech", // "max_segment" (seconds, > 0), "trim" (seconds >= 0, "segments" mode and the -// transcribe functions; 0 = no trimming), "mode" ("speech" or "segments"), +// transcribe functions; 0 = no trimming), "run_gate" (0 <= x < 1, default 0 = +// off; a speech run whose median probability is below it is dropped, in both +// modes and in the transcribe functions; not for streams), "mode" ("speech" or "segments"), // "probabilities" (bool), "speech_pad" (seconds >= 0, "speech" mode). With // `allow_filter` the word filter keys of parse_filter_options are accepted too // and stored in `req.filter`. Unknown keys and bad values are errors. The diff --git a/src/vad_segmenter.cpp b/src/vad_segmenter.cpp index bdd5212..e6f543e 100644 --- a/src/vad_segmenter.cpp +++ b/src/vad_segmenter.cpp @@ -10,6 +10,28 @@ namespace pk { namespace { // Above this many seconds a frame count could overflow int64. constexpr double kMaxSec = 1e6; + +// Run gate: every run of sp == 1 (a maximal run of frames with p >= threshold) +// whose median probability is below `gate` becomes silence. The median is over +// p[a, b) of the run; for an even count it is the mean of the two middle values, +// computed in float (as numpy does for float32). A run is kept when median >= gate. No-op for gate <= 0. +void apply_run_gate(std::vector& sp, const std::vector& p, int64_t n, float gate) { + if (!(gate > 0.0f)) return; + std::vector v; + int64_t f = 0; + while (f < n) { + if (!sp[(size_t)f]) { ++f; continue; } + int64_t e = f; + while (e < n && sp[(size_t)e]) ++e; + v.assign(p.begin() + f, p.begin() + e); // sp is only set where f < p.size() + const size_t m = v.size() / 2; + std::nth_element(v.begin(), v.begin() + m, v.end()); + float med = v[m]; + if (v.size() % 2 == 0) med = (*std::max_element(v.begin(), v.begin() + m) + med) * 0.5f; + if (med < gate) std::fill(sp.begin() + f, sp.begin() + e, 0); + f = e; + } +} } std::vector segment_by_vad(const std::vector& p, double total_sec, @@ -20,7 +42,7 @@ std::vector segment_by_vad(const std::vector& p, double total !(o.max_seg_sec > 2.0 * o.frame_sec) || !std::isfinite(total_sec) || !std::isfinite(o.threshold) || !std::isfinite(o.min_seg_sec) || !std::isfinite(o.min_pause_sec) || !std::isfinite(o.bridge_sec) || - !std::isfinite(o.min_speech_sec) || !std::isfinite(o.trim_sec) || o.max_seg_sec > kMaxSec || o.min_seg_sec > kMaxSec || + !std::isfinite(o.min_speech_sec) || !std::isfinite(o.trim_sec) || !std::isfinite(o.run_gate) || o.max_seg_sec > kMaxSec || o.min_seg_sec > kMaxSec || o.min_pause_sec > kMaxSec || o.bridge_sec > kMaxSec || o.min_speech_sec > kMaxSec) { out.push_back({0.0, total_sec}); return out; @@ -38,6 +60,7 @@ std::vector segment_by_vad(const std::vector& p, double total // Step 1: speech mask over n frames (frames without a probability are silent). std::vector sp((size_t)n, 0); for (int64_t f = 0; f < n && f < (int64_t)p.size(); ++f) sp[(size_t)f] = p[(size_t)f] >= o.threshold; + apply_run_gate(sp, p, n, o.run_gate); // Runs of equal value as [begin, end) frame ranges. auto for_runs = [&](bool value, auto&& fn) { int64_t f = 0; @@ -106,7 +129,7 @@ std::vector speech_regions(const std::vector& p, double total if (!(o.frame_sec > 0.0) || !std::isfinite(o.frame_sec) || !std::isfinite(total_sec) || !std::isfinite(o.threshold) || !std::isfinite(o.min_pause_sec) || !std::isfinite(o.bridge_sec) || !std::isfinite(o.min_speech_sec) || - !std::isfinite(o.pad_sec) || o.pad_sec < 0.0 || o.pad_sec > kMaxSec || + !std::isfinite(o.pad_sec) || o.pad_sec < 0.0 || o.pad_sec > kMaxSec || !std::isfinite(o.run_gate) || o.min_pause_sec > kMaxSec || o.bridge_sec > kMaxSec || o.min_speech_sec > kMaxSec) { out.push_back({0.0, total_sec}); return out; @@ -116,6 +139,7 @@ std::vector speech_regions(const std::vector& p, double total const int64_t pause_f = std::max(1, (int64_t)std::ceil(o.min_pause_sec / fs - 1e-9)); std::vector sp((size_t)n, 0); for (int64_t f = 0; f < n && f < (int64_t)p.size(); ++f) sp[(size_t)f] = p[(size_t)f] >= o.threshold; + apply_run_gate(sp, p, n, o.run_gate); auto for_runs = [&](bool value, auto&& fn) { int64_t f = 0; while (f < n) { diff --git a/src/vad_segmenter.hpp b/src/vad_segmenter.hpp index a9842bf..c5f6805 100644 --- a/src/vad_segmenter.hpp +++ b/src/vad_segmenter.hpp @@ -23,6 +23,17 @@ struct SegmenterOpts { // whole cut (the behaviour before trimming existed). Audio of at most // max_seg_sec is not affected. double trim_sec = 0.3; + // Run gate, off at 0. A speech run (a maximal stretch of frames with + // p >= threshold, before bridging) is dropped when the median of the + // probabilities of its frames is below this value, a probability in [0, 1). + // A median equal to the gate keeps the run. For an even number of frames + // the median is the mean of the two middle values. The gate acts first: + // dropped frames are silence for bridging, min_speech, pauses and trim. It + // applies to segment_by_vad and speech_regions, not to VadEventTracker. It + // is meant for the Ultra/Redux heads, whose noise runs have a lower median + // than speech runs (docs/vad.md). Any value <= 0 is off; NaN or infinity is + // a degenerate option. + float run_gate = 0.0f; }; // Which model made the probabilities. The two kinds differ in frame period @@ -47,7 +58,8 @@ SegmenterOpts default_segmenter_opts(VadKind kind); // Cuts [0, total_sec] into segments of at most max_seg_sec, at pauses found in // the per-frame speech probabilities p. // -// 1. A frame is speech when p >= threshold. Speech gaps shorter than bridge_sec +// 1. A frame is speech when p >= threshold. With run_gate > 0, a run of speech +// frames whose median probability is below run_gate becomes silence. Speech gaps shorter than bridge_sec // are filled, then speech runs shorter than min_speech_sec are removed. The // remaining silent runs of at least min_pause_sec are the pauses. // 2. Audio of at most max_seg_sec is returned whole, with or without speech. @@ -66,7 +78,7 @@ SegmenterOpts default_segmenter_opts(VadKind kind); // at least that run. Trimmed edges are not on the frame grid. // // Degenerate options (frame_sec not finite or <= 0, max_seg_sec not finite or -// <= 2 * frame_sec, threshold, trim_sec or any of the four durations not finite, +// <= 2 * frame_sec, threshold, run_gate, trim_sec or any of the four durations not finite, // any duration above 1e6 seconds) return the single segment {0, total_sec}. // Every cut between two segments is a whole number of frames. std::vector segment_by_vad(const std::vector& p, double total_sec, @@ -74,7 +86,7 @@ std::vector segment_by_vad(const std::vector& p, double total // The speech regions themselves, for any audio length (no cap, no cuts). // Uses threshold, frame_sec, bridge_sec, min_speech_sec and min_pause_sec: the -// speech mask is smoothed as in segment_by_vad (bridge gaps shorter than +// speech mask is gated (run_gate) and smoothed as in segment_by_vad (bridge gaps shorter than // bridge_sec, drop runs shorter than min_speech_sec), then speech runs // separated by a gap shorter than min_pause_sec are merged. Regions are // ordered, disjoint and inside [0, total_sec]; the result is empty when there diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 587e4f9..05e73aa 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -18,6 +18,7 @@ pk_add_test(test_model_loader_ternary) pk_add_test(test_ternary_load_negative) pk_add_test(test_vad_head) pk_add_test(test_vad_segmenter) +pk_add_test(test_vad_run_gate) pk_add_test(test_word_filter) pk_add_test(test_fft) pk_add_test(test_mel) @@ -254,11 +255,13 @@ pk_add_test(test_capi_vad_silero) pk_add_test(test_transcribe_vad_silero) pk_add_test(test_vad_only) pk_add_test(test_vad_trim_filter) +pk_add_test(test_vad_run_gate_model) target_compile_definitions(test_silero_vad PRIVATE PK_SOURCE_DIR="${CMAKE_SOURCE_DIR}") set_tests_properties(test_silero_vad test_capi_vad_silero test_transcribe_vad_silero PROPERTIES LABELS "model") set_tests_properties(test_capi_vad_silero test_transcribe_vad_silero PROPERTIES WORKING_DIRECTORY ${CMAKE_SOURCE_DIR}) set_tests_properties(test_vad_only PROPERTIES LABELS "model") set_tests_properties(test_vad_trim_filter PROPERTIES LABELS "model" WORKING_DIRECTORY ${CMAKE_SOURCE_DIR}) +set_tests_properties(test_vad_run_gate_model PROPERTIES LABELS "model" WORKING_DIRECTORY ${CMAKE_SOURCE_DIR}) pk_add_test(test_bundle) pk_add_test(test_bundle_models) @@ -279,6 +282,16 @@ if(TARGET parakeet-cli) pk_cli_flag_test(cli_min_local_conf_nan "--min-local-conf must be in" --min-local-conf abc) pk_cli_flag_test(cli_local_radius_zero "--local-radius must be > 0" --local-radius 0) pk_cli_flag_test(cli_vad_trim_negative "--vad-trim must be >= 0" --vad-trim -1) + pk_cli_flag_test(cli_vad_run_gate_negative "--vad-run-gate must be in" --vad-run-gate -0.1) + pk_cli_flag_test(cli_vad_run_gate_one "--vad-run-gate must be in" --vad-run-gate 1) + pk_cli_flag_test(cli_vad_run_gate_text "--vad-run-gate must be in" --vad-run-gate abc) + function(pk_cli_vad_flag_test name pattern) + add_test(NAME ${name} COMMAND parakeet-cli vad --model none.gguf --input none.wav ${ARGN}) + set_tests_properties(${name} PROPERTIES PASS_REGULAR_EXPRESSION "${pattern}") + endfunction() + pk_cli_vad_flag_test(cli_vad_cmd_run_gate_negative "--run-gate must be in" --run-gate -1) + pk_cli_vad_flag_test(cli_vad_cmd_run_gate_one "--run-gate must be in" --run-gate 1.0) + pk_cli_vad_flag_test(cli_vad_cmd_run_gate_text "--run-gate must be in" --run-gate x) pk_cli_flag_test(cli_filter_with_beam "word filter works with greedy decoding only" --min-local-conf 0.5 --beam-size 4) pk_cli_flag_test(cli_filter_with_stream "word filter is offline only" --drop-punct-only --stream) endif() diff --git a/tests/test_capi_vad_silero.cpp b/tests/test_capi_vad_silero.cpp index a580072..9b829c8 100644 --- a/tests/test_capi_vad_silero.cpp +++ b/tests/test_capi_vad_silero.cpp @@ -242,6 +242,14 @@ int main() { CHECK(std::strlen(parakeet_capi_last_error(ctx)) > 0); CHECK(parakeet_capi_vad_stream_begin(ctx, 16000, "{\"mode\":\"segments\"}") == nullptr); CHECK(parakeet_capi_vad_stream_begin(ctx, 16000, "{\"bogus\":1}") == nullptr); + // run_gate is offline only: a stream refuses it, and run_gate 0 is fine. + CHECK(parakeet_capi_vad_stream_begin(ctx, 16000, "{\"run_gate\":0.9}") == nullptr); + CHECK(std::string(parakeet_capi_last_error(ctx)).find("run_gate") != std::string::npos); + { + parakeet_vad_stream* z = parakeet_capi_vad_stream_begin(ctx, 16000, "{\"run_gate\":0}"); + CHECK(z != nullptr); + parakeet_capi_vad_stream_free(z); + } auto run = [&](parakeet_vad_stream* st, unsigned seed, std::vector* probs, std::vector>* ev) { diff --git a/tests/test_vad_options.cpp b/tests/test_vad_options.cpp index b7b894c..6dd2e14 100644 --- a/tests/test_vad_options.cpp +++ b/tests/test_vad_options.cpp @@ -76,6 +76,30 @@ int main() { CHECK(!parse("{\"trim\":1e9}", k, r, err) && err.find("trim") != std::string::npos); } + // "run_gate": off (0) by default for both kinds, a number in [0, 1), else an error. + CHECK(parse(nullptr, VadKind::kHead, r, err) && r.opts.run_gate == 0.0f); + CHECK(parse(nullptr, VadKind::kSilero, r, err) && r.opts.run_gate == 0.0f); + CHECK(parse("{\"run_gate\":0.92}", VadKind::kHead, r, err) && std::fabs(r.opts.run_gate - 0.92f) < 1e-6 && + near(r.opts.trim_sec, 0.3) && near(r.opts.min_pause_sec, 0.2)); + CHECK(parse("{\"run_gate\":0.5,\"mode\":\"segments\"}", VadKind::kSilero, r, err) && r.opts.run_gate == 0.5f && + r.mode == VadRequest::Mode::kSegments); + CHECK(parse("{\"run_gate\":0}", VadKind::kHead, r, err) && r.opts.run_gate == 0.0f); + CHECK(parse("{\"run_gate\":0.9999}", VadKind::kHead, r, err) && r.opts.run_gate < 1.0f); + for (VadKind k : {VadKind::kHead, VadKind::kSilero}) { + for (const char* j : {"{\"run_gate\":-0.1}", "{\"run_gate\":1}", "{\"run_gate\":1.5}", "{\"run_gate\":\"x\"}", + "{\"run_gate\":true}", "{\"run_gate\":1e400}", "{\"run_gate\":}"}) { + CHECK(!parse(j, k, r, err) && err.find("run_gate") != std::string::npos); + } + } + CHECK(!parse("{\"run_gates\":0.5}", VadKind::kHead, r, err) && err.find("unknown") != std::string::npos); + // The word filter keys parser takes no run_gate; the transcribe parser with the filter does. + CHECK(parse_vad_options("{\"run_gate\":0.9,\"min_local_conf\":0.5}", r, err, VadKind::kHead, true) && + std::fabs(r.opts.run_gate - 0.9f) < 1e-6 && r.filter.active()); + { + WordFilter f; + CHECK(!parse_filter_options("{\"run_gate\":0.9}", f, err) && err.find("run_gate") != std::string::npos); + } + // Word filter keys: only with allow_filter; off by default. CHECK(parse(nullptr, VadKind::kHead, r, err) && !r.filter.active() && near(r.filter.local_radius_sec, 5.0)); CHECK(!parse_vad_options("{\"min_local_conf\":0.5}", r, err, VadKind::kHead, false) && diff --git a/tests/test_vad_run_gate.cpp b/tests/test_vad_run_gate.cpp new file mode 100644 index 0000000..ce5ea20 --- /dev/null +++ b/tests/test_vad_run_gate.cpp @@ -0,0 +1,313 @@ +// The opt-in run gate of the segmenter (SegmenterOpts::run_gate): a speech run +// is dropped when the median of its frame probabilities is below the gate. +// Model-independent: synthetic probability streams with exact expected output, +// for the 0.08 s head frame and the 0.032 s Silero frame. +// +// The definition under test: +// * a run is a maximal stretch of frames with p >= threshold (before any +// bridging), so a bridged gap is never inside a run; +// * the median is over the probabilities of the frames of that run (odd +// count: the middle value; even count: the mean of the two middle values); +// * the run is kept when median >= run_gate, dropped when it is below; +// * dropped frames count as silence for bridging, the min_speech rule, pauses +// and trim (the gate runs first); +// * run_gate 0 is off and gives the output of the segmenter before the gate. +#include "vad_segmenter.hpp" + +#include +#include +#include +#include +#include +#include +#include + +using namespace pk; + +static int failures = 0; +#define CHECK(c) do { if (!(c)) { std::fprintf(stderr, "FAIL: %s (line %d)\n", #c, __LINE__); ++failures; } } while (0) + +static bool near(double a, double b) { return std::fabs(a - b) < 1e-6; } + +struct Fill { int a, b; float v; }; // frames [a, b) at probability v + +// n frames at p = 0.02, then the fills in order (a later fill overwrites). +static std::vector stream(int n, std::initializer_list fills) { + std::vector p((size_t)n, 0.02f); + for (const Fill& f : fills) + for (int i = f.a; i < f.b && i < n; ++i) p[(size_t)i] = f.v; + return p; +} + +static bool seg_is(const std::vector& s, std::initializer_list> want) { + if (s.size() != want.size()) return false; + size_t i = 0; + for (auto w : want) { + if (!near(s[i].start, w.first) || !near(s[i].end, w.second)) return false; + ++i; + } + return true; +} + +static bool same_segs(const std::vector& a, const std::vector& b) { + if (a.size() != b.size()) return false; + for (size_t i = 0; i < a.size(); ++i) + if (a[i].start != b[i].start || a[i].end != b[i].end) return false; // exact: byte for byte + return true; +} + +static SegmenterOpts gated(SegmenterOpts o, float g) { + o.run_gate = g; + return o; +} + +static void test_default_is_off() { + CHECK(SegmenterOpts().run_gate == 0.0f); + CHECK(default_segmenter_opts(VadKind::kHead).run_gate == 0.0f); + CHECK(default_segmenter_opts(VadKind::kSilero).run_gate == 0.0f); +} + +// Head frames. Runs: A [10,30) at 0.97, B [50,70) at 0.80, C [100,110) at 0.6 +// with a single peak of 0.99 (low median, high peak), D one frame at 0.99. +static void test_head_speech_regions() { + SegmenterOpts o; + o.min_speech_sec = 0.05; // a one-frame run (0.08 s) is speech + const double total = 15.0; + auto p = stream(188, {{10, 30, 0.97f}, {50, 70, 0.80f}, {100, 110, 0.6f}, {105, 106, 0.99f}, {130, 131, 0.99f}}); + // Off: all four runs. + CHECK(seg_is(speech_regions(p, total, o), {{0.8, 2.4}, {4.0, 5.6}, {8.0, 8.8}, {10.4, 10.48}})); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.0f)), {{0.8, 2.4}, {4.0, 5.6}, {8.0, 8.8}, {10.4, 10.48}})); + // 0.7: C goes although its peak is 0.99 (its median is 0.6); B stays. + CHECK(seg_is(speech_regions(p, total, gated(o, 0.7f)), {{0.8, 2.4}, {4.0, 5.6}, {10.4, 10.48}})); + // 0.9: B goes as well. The single frame at 0.99 has median 0.99 and stays. + CHECK(seg_is(speech_regions(p, total, gated(o, 0.9f)), {{0.8, 2.4}, {10.4, 10.48}})); + // 0.995: nothing is left. + CHECK(speech_regions(p, total, gated(o, 0.995f)).empty()); +} + +// Boundary: median exactly equal to the gate keeps the run (>=). +static void test_boundary() { + SegmenterOpts o; + const double total = 8.0; + // Odd run: 3 frames at 0.75 (exact in float): median 0.75. + auto p = stream(100, {{10, 13, 0.75f}}); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.75f)), {{0.8, 1.04}})); + CHECK(speech_regions(p, total, gated(o, 0.76f)).empty()); + // Even run, values not in order: {0.875, 0.5, 1.0, 0.625} has median + // (0.625 + 0.875) / 2 = 0.75 and mean 0.75. + p = stream(100, {{10, 11, 0.875f}, {11, 12, 0.5f}, {12, 13, 1.0f}, {13, 14, 0.625f}}); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.75f)), {{0.8, 1.12}})); + CHECK(speech_regions(p, total, gated(o, 0.76f)).empty()); + // {0.5, 0.5, 0.875, 1.0}: median 0.6875, mean 0.719. The median decides. + p = stream(100, {{10, 12, 0.5f}, {12, 13, 0.875f}, {13, 14, 1.0f}}); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.6875f)), {{0.8, 1.12}})); + CHECK(speech_regions(p, total, gated(o, 0.7f)).empty()); + // A frame exactly at the threshold belongs to the run: {0.5, 0.5, 0.5}. + p = stream(100, {{10, 13, 0.5f}}); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.5f)), {{0.8, 1.04}})); + CHECK(speech_regions(p, total, gated(o, 0.51f)).empty()); +} + +// The run is the raw run of p >= threshold, not the bridged one. +static void test_bridged_gap() { + SegmenterOpts o; // bridge 0.1 s bridges a one-frame gap (0.08 s) at 0.08 s frames + const double total = 8.0; + // Two good runs and a one-frame gap: bridged into one region, gate keeps it. + auto p = stream(100, {{10, 20, 0.97f}, {21, 31, 0.97f}}); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.9f)), {{0.8, 2.48}})); + CHECK(seg_is(speech_regions(p, total, o), {{0.8, 2.48}})); + // Run 1 is weak (0.6 x 3), run 2 is strong (0.97 x 20), one frame between. + // The bridged run would have median 0.97 over its 24 frames; the raw run 1 + // has median 0.6 and goes. Only run 2 is left. + p = stream(100, {{10, 13, 0.6f}, {14, 34, 0.97f}}); + CHECK(seg_is(speech_regions(p, total, o), {{0.8, 2.72}})); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.9f)), {{1.12, 2.72}})); + // Run 1 strong, run 2 weak after a bridgeable gap: only run 1 stays, the + // bridged gap does not come back. + p = stream(100, {{10, 20, 0.97f}, {21, 24, 0.6f}}); + CHECK(seg_is(speech_regions(p, total, o), {{0.8, 1.92}})); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.9f)), {{0.8, 1.6}})); + // Dropping a run that sat between two kept runs opens a gap that is no + // longer bridged: kept 10 frames, weak 1 frame ... gap 1 + weak run 2 + gap 1. + p = stream(100, {{10, 20, 0.97f}, {21, 23, 0.6f}, {24, 34, 0.97f}}); + CHECK(seg_is(speech_regions(p, total, o), {{0.8, 2.72}})); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.9f)), {{0.8, 1.6}, {1.92, 2.72}})); +} + +// A gated-out run is silence for the min_speech rule too: a short run next to a +// dropped one is not merged with it. +static void test_min_speech_after_gate() { + SegmenterOpts o; // min_speech 0.1 s: one frame (0.08 s) is too short, two are enough + const double total = 8.0; + auto p = stream(100, {{10, 12, 0.99f}, {20, 21, 0.99f}}); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.9f)), {{0.8, 0.96}})); + // The single frame is not rescued by the gate; it fails min_speech first or + // after, with the same result. + p = stream(100, {{10, 11, 0.99f}}); + CHECK(speech_regions(p, total, gated(o, 0.9f)).empty()); +} + +// segment_by_vad on 45 s of head frames (563 frames): speech [20,40) at 0.97 and +// a noise-like run at 0.6. +static void test_head_segments_and_trim() { + const SegmenterOpts o; // trim 0.3 + auto p = stream(563, {{20, 40, 0.97f}, {60, 80, 0.6f}}); + CHECK(seg_is(segment_by_vad(p, 45.0, o), {{1.6 - 0.3, 3.2 + 0.3}, {4.8 - 0.3, 6.4 + 0.3}})); + CHECK(seg_is(segment_by_vad(p, 45.0, gated(o, 0.0f)), {{1.6 - 0.3, 3.2 + 0.3}, {4.8 - 0.3, 6.4 + 0.3}})); + // The gate drops the second run; the cut stays at the middle of the long + // silence and the piece is trimmed to the one run that is left. + CHECK(seg_is(segment_by_vad(p, 45.0, gated(o, 0.9f)), {{1.6 - 0.3, 3.2 + 0.3}})); + // Everything weak: nothing is left. + CHECK(segment_by_vad(p, 45.0, gated(o, 0.99f)).empty()); + + // Gate first, then trim: a weak run 0.16 s after the strong one is inside + // the trim margin without the gate and does not count with it. + p = stream(563, {{20, 40, 0.97f}, {42, 46, 0.6f}}); + CHECK(seg_is(segment_by_vad(p, 45.0, o), {{1.6 - 0.3, 46 * 0.08 + 0.3}})); + CHECK(seg_is(segment_by_vad(p, 45.0, gated(o, 0.9f)), {{1.6 - 0.3, 3.2 + 0.3}})); + // With trim 0 the cut itself is unchanged by the gate (only the empty piece + // is dropped): cut at the middle of the silence after the run. + SegmenterOpts t0 = o; + t0.trim_sec = 0.0; + CHECK(seg_is(segment_by_vad(p, 45.0, gated(t0, 0.9f)), {{0.0, 301 * 0.08}})); + + // Audio of at most max_seg_sec is returned whole, with or without the gate. + p = stream(300, {{20, 40, 0.6f}}); + CHECK(seg_is(segment_by_vad(p, 24.0, gated(o, 0.9f)), {{0.0, 24.0}})); + // A gated-out hard-cut piece: speech through the hard cut stays one run. + p = stream(875, {{0, 875, 0.97f}}); + CHECK(seg_is(segment_by_vad(p, 70.0, gated(o, 0.9f)), {{0.0, 30.0}, {30.0, 60.0}, {60.0, 70.0}})); + p = stream(875, {{0, 875, 0.6f}}); + CHECK(segment_by_vad(p, 70.0, gated(o, 0.9f)).empty()); +} + +// Silero frames (0.032 s), default Silero options but no padding. +static void test_silero() { + SegmenterOpts o = default_segmenter_opts(VadKind::kSilero); + o.pad_sec = 0.0; + const double fs = 0.032; + // A [100,140) 0.97, B [200,240) 0.8, C [300,340) 0.6 with one 0.99 peak. + auto p = stream(500, {{100, 140, 0.97f}, {200, 240, 0.8f}, {300, 340, 0.6f}, {320, 321, 0.99f}}); + const double total = 16.0; + CHECK(seg_is(speech_regions(p, total, o), {{100 * fs, 140 * fs}, {200 * fs, 240 * fs}, {300 * fs, 340 * fs}})); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.7f)), {{100 * fs, 140 * fs}, {200 * fs, 240 * fs}})); + CHECK(seg_is(speech_regions(p, total, gated(o, 0.9f)), {{100 * fs, 140 * fs}})); + // With Silero's own padding the kept region is padded as before. + SegmenterOpts padded = default_segmenter_opts(VadKind::kSilero); + CHECK(seg_is(speech_regions(p, total, gated(padded, 0.9f)), {{100 * fs - 0.03, 140 * fs + 0.03}})); + // A run just under min_speech (7 frames = 0.224 s) is dropped with or + // without the gate; one of 8 frames (0.256 s) is kept without it and dropped + // by it when weak. + p = stream(500, {{100, 107, 0.99f}, {200, 208, 0.6f}}); + CHECK(seg_is(speech_regions(p, total, o), {{200 * fs, 208 * fs}})); + CHECK(speech_regions(p, total, gated(o, 0.9f)).empty()); + + // Segments, 45 s = 1407 frames: runs [900,937) and [1200,1250). The second + // is weak. The first cut lands in the leading silence (its piece has no + // speech), the second in the pause after the strong run. + p = stream(1407, {{900, 937, 0.97f}, {1200, 1250, 0.8f}}); + const SegmenterOpts so = default_segmenter_opts(VadKind::kSilero); + CHECK(seg_is(segment_by_vad(p, 45.0, so), {{900 * fs - 0.3, 937 * fs + 0.3}, {1200 * fs - 0.3, 1250 * fs + 0.3}})); + CHECK(seg_is(segment_by_vad(p, 45.0, gated(so, 0.9f)), {{900 * fs - 0.3, 937 * fs + 0.3}})); + // The same stream gated at 0.8 keeps both (median 0.8 >= 0.8). + CHECK(seg_is(segment_by_vad(p, 45.0, gated(so, 0.8f)), {{900 * fs - 0.3, 937 * fs + 0.3}, {1200 * fs - 0.3, 1250 * fs + 0.3}})); +} + +static void test_degenerate_gate() { + SegmenterOpts o; + auto p = stream(563, {{20, 40, 0.97f}}); + SegmenterOpts bad = o; + bad.run_gate = NAN; + CHECK(seg_is(segment_by_vad(p, 45.0, bad), {{0.0, 45.0}})); + CHECK(seg_is(speech_regions(p, 45.0, bad), {{0.0, 45.0}})); + bad.run_gate = INFINITY; + CHECK(seg_is(segment_by_vad(p, 45.0, bad), {{0.0, 45.0}})); + // A negative gate is off. + bad.run_gate = -0.5f; + CHECK(same_segs(segment_by_vad(p, 45.0, bad), segment_by_vad(p, 45.0, o))); + CHECK(same_segs(speech_regions(p, 45.0, bad), speech_regions(p, 45.0, o))); +} + +// Random streams with random probabilities. +static std::vector random_stream(std::mt19937& rng, int n) { + std::vector p; + bool sp = rng() & 1; + while ((int)p.size() < n) { + const int len = 1 + (int)(rng() % 60); + for (int i = 0; i < len && (int)p.size() < n; ++i) { + const float u = (float)(rng() % 1000) / 1000.0f; + p.push_back(sp ? 0.5f + 0.5f * u : 0.5f * u * 0.99f); + } + sp = !sp; + } + return p; +} + +static void test_random_properties() { + std::mt19937 rng(20261006); + for (int trial = 0; trial < 300; ++trial) { + SegmenterOpts o = default_segmenter_opts(trial % 2 ? VadKind::kSilero : VadKind::kHead); + o.pad_sec = 0.0; + o.trim_sec = (trial % 3) * 0.1; + const int n = (int)(35.0 / o.frame_sec) + (int)(rng() % 3000); + const auto p = random_stream(rng, n); + const double total = (double)n * o.frame_sec; + // A gate no run can fail (every frame of a run is >= 0.5) changes nothing. + CHECK(same_segs(segment_by_vad(p, total, gated(o, 0.4f)), segment_by_vad(p, total, o))); + CHECK(same_segs(speech_regions(p, total, gated(o, 0.4f)), speech_regions(p, total, o))); + CHECK(same_segs(speech_regions(p, total, gated(o, 1e-9f)), speech_regions(p, total, o))); + // A gate only removes speech: every gated region lies inside a region of + // the ungated output (padding off). + const float g = 0.55f + 0.04f * (float)(trial % 10); + const auto all = speech_regions(p, total, o); + const auto some = speech_regions(p, total, gated(o, g)); + for (const VadSegment& s : some) { + bool inside = false; + for (const VadSegment& a : all) + if (s.start >= a.start - 1e-9 && s.end <= a.end + 1e-9) inside = true; + CHECK(inside); + } + // A higher gate never keeps more speech. + double dur_lo = 0.0, dur_hi = 0.0; + for (const VadSegment& s : speech_regions(p, total, gated(o, g))) dur_lo += s.end - s.start; + for (const VadSegment& s : speech_regions(p, total, gated(o, g + 0.05f))) dur_hi += s.end - s.start; + CHECK(dur_hi <= dur_lo + 1e-9); + // Segments: ordered, disjoint, each inside the audio. + double prev = 0.0; + for (const VadSegment& s : segment_by_vad(p, total, gated(o, g))) { + CHECK(s.start >= prev - 1e-9 && s.end > s.start && s.end <= total + 1e-9); + prev = s.end; + } + } +} + +// The event tracker (streaming) has no run median: the gate does not change it. +static void test_event_tracker_ignores_gate() { + SegmenterOpts o = default_segmenter_opts(VadKind::kSilero); + SegmenterOpts g = gated(o, 0.95f); + auto p = stream(500, {{100, 140, 0.6f}, {200, 240, 0.97f}}); + VadEventTracker a(o), b(g); + std::vector ea, eb; + for (float v : p) { a.push(v, &ea); b.push(v, &eb); } + a.finish(16.0, &ea); + b.finish(16.0, &eb); + CHECK(ea.size() == eb.size() && ea.size() == 4); + for (size_t i = 0; i < ea.size() && i < eb.size(); ++i) + CHECK(ea[i].start == eb[i].start && ea[i].time == eb[i].time); +} + +int main() { + test_default_is_off(); + test_head_speech_regions(); + test_boundary(); + test_bridged_gap(); + test_min_speech_after_gate(); + test_head_segments_and_trim(); + test_silero(); + test_degenerate_gate(); + test_random_properties(); + test_event_tracker_ignores_gate(); + if (failures) { std::fprintf(stderr, "%d failure(s)\n", failures); return 1; } + std::printf("test_vad_run_gate: OK\n"); + return 0; +} diff --git a/tests/test_vad_run_gate_model.cpp b/tests/test_vad_run_gate_model.cpp new file mode 100644 index 0000000..c171ef9 --- /dev/null +++ b/tests/test_vad_run_gate_model.cpp @@ -0,0 +1,271 @@ +// The VAD run gate on real detectors. The clip is two_speakers.wav, 1 s of +// digital silence, 12 s of seeded noise, 1 s of silence and speech.wav, so it +// has a noise stretch between two stretches of speech and is longer than the +// 30 s segment cap. Checked per detector (white and pink noise): +// * "run_gate":0 and no option give the same JSON, byte for byte, in both +// modes; +// * in "speech" mode a gate never adds speech; +// * with the Redux head at 0.92 most of the seconds the head called speech +// inside the noise stretch are gone, and the speech stretches are unchanged; +// * with a full ASR head the transcribe path honours the key: no word in the +// noise stretch, and "run_gate":0 gives the document of no option. +// Env (each optional; skip 77 when none is set): +// PARAKEET_TEST_VAD_ONLY_REDUX_GGUF a Redux VAD-only slice (scripts/slice_vad_gguf.py) +// PARAKEET_TEST_GGUF_REDUX_KEEP / _REDUX_DEQ / _ULTRA ASR models with a head +// PARAKEET_TEST_SILERO_GGUF a Silero VAD model +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "audio_io.hpp" +#include "model.hpp" +#include "parakeet_capi.h" + +using namespace pk; + +static int failures = 0; +#define CHECK(c) do { if (!(c)) { std::fprintf(stderr, "FAIL: %s (line %d)\n", #c, __LINE__); ++failures; } } while (0) + +struct Seg { double a, b; }; + +static std::string take(char* p) { + if (!p) return std::string(); + std::string s(p); + parakeet_capi_free_string(p); + return s; +} + +// The "segments" array of a VAD JSON document. +static std::vector segs_of(const std::string& j) { + std::vector out; + size_t i = j.find("\"segments\":["); + if (i == std::string::npos) return out; + const size_t end = j.find(']', i); + for (size_t k = j.find("\"start\":", i); k != std::string::npos && k < end; k = j.find("\"start\":", k + 1)) { + const double a = std::atof(j.c_str() + k + 8); + const size_t e = j.find("\"end\":", k); + out.push_back({a, std::atof(j.c_str() + e + 6)}); + } + return out; +} + +static double overlap(const std::vector& s, double lo, double hi) { + double t = 0.0; + for (const Seg& g : s) t += std::max(0.0, std::min(g.b, hi) - std::max(g.a, lo)); + return t; +} + +static double covered(const std::vector& s) { return overlap(s, -1e9, 1e9); } + +static bool same(const std::vector& a, const std::vector& b) { + if (a.size() != b.size()) return false; + for (size_t i = 0; i < a.size(); ++i) + if (std::fabs(a[i].a - b[i].a) > 1e-9 || std::fabs(a[i].b - b[i].b) > 1e-9) return false; + return true; +} + +static std::vector within(const std::vector& s, double lo, double hi) { + std::vector o; + for (const Seg& g : s) if (g.a >= lo && g.b <= hi) o.push_back(g); + return o; +} + +// White noise, or pink noise from Paul Kellet's filter, scaled to the given RMS. +static std::vector make_noise(bool pink, size_t n, float rms, unsigned seed) { + std::mt19937 rng(seed); + std::normal_distribution nd(0.0f, 1.0f); + std::vector x(n); + float b0 = 0, b1 = 0, b2 = 0; + for (size_t i = 0; i < n; ++i) { + const float w = nd(rng); + if (pink) { + b0 = 0.99765f * b0 + w * 0.0990460f; + b1 = 0.96300f * b1 + w * 0.2965164f; + b2 = 0.57000f * b2 + w * 1.0526913f; + x[i] = b0 + b1 + b2 + w * 0.1848f; + } else { + x[i] = w; + } + } + double e = 0.0; + for (float v : x) e += (double)v * v; + const float g = rms / (float)std::sqrt(e / (double)n + 1e-30); + for (float& v : x) v *= g; + return x; +} + +static bool write_wav16(const std::string& path, const std::vector& pcm) { + std::FILE* f = std::fopen(path.c_str(), "wb"); + if (!f) return false; + const uint32_t n = (uint32_t)pcm.size() * 2, sr = 16000, br = sr * 2, fmt = 16, riff = 36 + n; + const uint16_t pcm_tag = 1, ch = 1, align = 2, bits = 16; + std::fwrite("RIFF", 1, 4, f); std::fwrite(&riff, 4, 1, f); std::fwrite("WAVEfmt ", 1, 8, f); + std::fwrite(&fmt, 4, 1, f); std::fwrite(&pcm_tag, 2, 1, f); std::fwrite(&ch, 2, 1, f); + std::fwrite(&sr, 4, 1, f); std::fwrite(&br, 4, 1, f); std::fwrite(&align, 2, 1, f); std::fwrite(&bits, 2, 1, f); + std::fwrite("data", 1, 4, f); std::fwrite(&n, 4, 1, f); + for (float v : pcm) { + const int16_t s = (int16_t)std::lrintf(std::max(-1.0f, std::min(1.0f, v)) * 32767.0f); + std::fwrite(&s, 2, 1, f); + } + std::fclose(f); + return true; +} + +struct Clip { + std::vector pcm; + double speech_a_end = 0, noise_lo = 0, noise_hi = 0, speech_b_start = 0, total = 0; +}; + +static Clip make_clip(bool pink) { + Audio a, b; + CHECK(load_audio_16k_mono("tests/fixtures/two_speakers.wav", a)); + CHECK(load_audio_16k_mono("tests/fixtures/speech.wav", b)); + Clip c; + c.pcm = a.samples; + c.speech_a_end = (double)c.pcm.size() / 16000.0; + c.pcm.insert(c.pcm.end(), 16000, 0.0f); + const std::vector z = make_noise(pink, 12 * 16000, 0.08f, pink ? 11u : 7u); + c.noise_lo = (double)c.pcm.size() / 16000.0; + c.pcm.insert(c.pcm.end(), z.begin(), z.end()); + c.noise_hi = (double)c.pcm.size() / 16000.0; + c.pcm.insert(c.pcm.end(), 16000, 0.0f); + c.speech_b_start = (double)c.pcm.size() / 16000.0; + c.pcm.insert(c.pcm.end(), b.samples.begin(), b.samples.end()); + c.total = (double)c.pcm.size() / 16000.0; + return c; +} + +static std::string vad(parakeet_ctx* ctx, const Clip& c, const char* opts) { + return take(parakeet_capi_vad_pcm_json(ctx, c.pcm.data(), (int)c.pcm.size(), 16000, opts)); +} + +// `redux`: the head is the Redux head, so the gate must remove the noise. +static void check_detector(const char* path, const char* what, bool redux, double gate) { + parakeet_ctx* ctx = parakeet_capi_load(path); + CHECK(ctx != nullptr); + if (!ctx) return; + for (int pink = 0; pink < 2; ++pink) { + const Clip c = make_clip(pink != 0); + std::fprintf(stderr, "%s, %s noise\n", what, pink ? "pink" : "white"); + char g[64], gj[128]; + std::snprintf(g, sizeof(g), "%.2f", gate); + for (const char* mode : {"speech", "segments"}) { + char o0[96], o1[96], o2[128]; + std::snprintf(o0, sizeof(o0), "{\"mode\":\"%s\"}", mode); + std::snprintf(o1, sizeof(o1), "{\"mode\":\"%s\",\"run_gate\":0}", mode); + std::snprintf(gj, sizeof(gj), "{\"mode\":\"%s\",\"run_gate\":%s}", mode, g); + std::snprintf(o2, sizeof(o2), "{\"mode\":\"%s\",\"run_gate\":%s,\"probabilities\":true}", mode, g); + const std::string off = vad(ctx, c, o0); + CHECK(!off.empty()); + CHECK(vad(ctx, c, o1) == off); // 0 is byte-identical to no key + if (std::strcmp(mode, "speech") == 0) CHECK(vad(ctx, c, nullptr) == off); + const std::string on = vad(ctx, c, gj); + CHECK(!on.empty()); + const auto s_off = segs_of(off), s_on = segs_of(on); + const double n_off = overlap(s_off, c.noise_lo + 0.5, c.noise_hi - 0.5); + const double n_on = overlap(s_on, c.noise_lo + 0.5, c.noise_hi - 0.5); + std::fprintf(stderr, " %-8s noise stretch called speech: %.2f s off, %.2f s gate %s; total %.2f -> %.2f s\n", + mode, n_off, n_on, g, covered(s_off), covered(s_on)); + // The probabilities in the document are the detector's own, gate or not. + const std::string withp = vad(ctx, c, o2); + CHECK(!withp.empty() && withp.find("\"probabilities\"") != std::string::npos); + if (std::strcmp(mode, "speech") == 0) { + CHECK(covered(s_on) <= covered(s_off) + 1e-6); // a gate never adds speech + CHECK(n_on <= n_off + 1e-6); + if (redux) { + CHECK(n_off > 2.0); // the noise does fool the head + CHECK(n_on <= 0.25 * n_off); // and the gate removes most of it + // The speech stretches keep their regions. + CHECK(same(within(s_off, 0.0, c.speech_a_end + 0.1), within(s_on, 0.0, c.speech_a_end + 0.1))); + CHECK(same(within(s_off, c.speech_b_start - 0.1, c.total), within(s_on, c.speech_b_start - 0.1, c.total))); + CHECK(!within(s_on, 0.0, c.speech_a_end + 0.1).empty()); + } + } else { + // The cuts follow the segmenter rules (last pause inside the 30 s + // window), so a long pause can stay inside a piece; the speech + // itself must be inside the pieces, with the gate on. + CHECK(!s_on.empty()); + CHECK(overlap(s_on, 0.6, c.speech_a_end - 0.2) > 0.9 * (c.speech_a_end - 0.8)); + CHECK(overlap(s_on, c.speech_b_start + 0.6, c.total - 0.4) > 0.9 * (c.total - c.speech_b_start - 1.0)); + } + } + // Bad value through the C-API. + CHECK(parakeet_capi_vad_pcm_json(ctx, c.pcm.data(), (int)c.pcm.size(), 16000, "{\"run_gate\":1}") == nullptr); + CHECK(std::string(parakeet_capi_last_error(ctx)).find("run_gate") != std::string::npos); + } + parakeet_capi_free(ctx); +} + +// A full ASR model with a head: the transcribe paths take the gate. +static void check_transcribe(const char* path, const char* what, double gate) { + std::fprintf(stderr, "%s: transcribe\n", what); + std::unique_ptr m = Model::load(path); + CHECK(m != nullptr); + if (!m || !m->config().vad.present) return; + const Clip c = make_clip(false); + SegmenterOpts o; + const Transcription base = m->transcribe_pcm_vad_with_timestamps(c.pcm, 16000, Decoder::kDefault, "", o); + SegmenterOpts z = o; + z.run_gate = 0.0f; + const Transcription zero = m->transcribe_pcm_vad_with_timestamps(c.pcm, 16000, Decoder::kDefault, "", z); + CHECK(zero.text == base.text && zero.words.size() == base.words.size()); + SegmenterOpts go = o; + go.run_gate = (float)gate; + const Transcription on = m->transcribe_pcm_vad_with_timestamps(c.pcm, 16000, Decoder::kDefault, "", go); + CHECK(!on.words.empty()); + size_t in_noise_on = 0, in_noise_off = 0; + for (const Word& w : on.words) in_noise_on += (w.start > c.noise_lo + 0.5 && w.start < c.noise_hi - 0.5); + for (const Word& w : base.words) in_noise_off += (w.start > c.noise_lo + 0.5 && w.start < c.noise_hi - 0.5); + std::fprintf(stderr, " words inside the noise stretch: %zu off, %zu gate %.2f; %zu / %zu words in all\n", + in_noise_off, in_noise_on, gate, base.words.size(), on.words.size()); + CHECK(in_noise_on == 0); + CHECK(on.words.size() <= base.words.size()); + CHECK(m->transcribe_pcm_vad(c.pcm, 16000, Decoder::kDefault, "", go) == on.text); + + // C-API: the key reaches the segmenter; 0 and no option give the same document. + const std::string wav = "test_vad_run_gate_model.tmp.wav"; + CHECK(write_wav16(wav, c.pcm)); + parakeet_ctx* ctx = parakeet_capi_load(path); + CHECK(ctx != nullptr); + if (ctx) { + const std::string a = take(parakeet_capi_transcribe_path_json_vad_with(ctx, nullptr, wav.c_str(), 0, nullptr)); + const std::string b = take(parakeet_capi_transcribe_path_json_vad_with(ctx, nullptr, wav.c_str(), 0, "{\"run_gate\":0}")); + const std::string d = take(parakeet_capi_transcribe_path_json_vad(ctx, wav.c_str(), 0)); + CHECK(!a.empty() && a == b && a == d); + char opts[64]; + std::snprintf(opts, sizeof(opts), "{\"run_gate\":%.2f}", gate); + const std::string e = take(parakeet_capi_transcribe_path_json_vad_with(ctx, nullptr, wav.c_str(), 0, opts)); + CHECK(!e.empty()); + CHECK(parakeet_capi_transcribe_path_json_vad_with(ctx, nullptr, wav.c_str(), 0, "{\"run_gate\":1.5}") == nullptr); + CHECK(std::string(parakeet_capi_last_error(ctx)).find("run_gate") != std::string::npos); + // Plain transcribe options have no segmenter: the key is unknown there. + CHECK(parakeet_capi_transcribe_path_json_with(ctx, wav.c_str(), 0, "{\"run_gate\":0.9}") == nullptr); + parakeet_capi_free(ctx); + } + std::remove(wav.c_str()); +} + +int main() { + const char* slice = std::getenv("PARAKEET_TEST_VAD_ONLY_REDUX_GGUF"); + const char* keep = std::getenv("PARAKEET_TEST_GGUF_REDUX_KEEP"); + const char* deq = std::getenv("PARAKEET_TEST_GGUF_REDUX_DEQ"); + const char* ultra = std::getenv("PARAKEET_TEST_GGUF_ULTRA"); + const char* silero = std::getenv("PARAKEET_TEST_SILERO_GGUF"); + if (!slice && !keep && !deq && !ultra && !silero) { std::puts("skip: no model env set"); return 77; } + if (slice) check_detector(slice, "Redux VAD slice", true, 0.92); + if (keep) { check_detector(keep, "Redux packed", true, 0.92); check_transcribe(keep, "Redux packed", 0.92); } + if (deq) check_detector(deq, "Redux dequantized", true, 0.92); + if (ultra) check_detector(ultra, "Ultra", false, 0.92); + if (silero) check_detector(silero, "Silero", false, 0.92); + if (failures) return 1; + std::puts("test_vad_run_gate_model: OK"); + return 0; +} From c09f7a2f2254fda1a5a1ef724145ae1cec3ac99e Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Tue, 6 Oct 2026 07:39:52 +0000 Subject: [PATCH 2/3] docs: the run gate, and the VAD study on real recordings Describe the run gate in docs/vad.md: the definition of the run median, how to use it, what it costs and what it does not do (music still triggers a head, and it is not a noise rejector). Add a Real recordings section to docs/vad-benchmarks.md: 59 recordings with human labels and 2.6 h without speech. It reports frame F1 untuned and tuned, the reference noise and a collar test, false alarms on audio without speech, the audio the decoder receives, word error rate on clean talks and on composites with inserted music and noise, the cost of the 30 s hard cut, and the limits. Update the guidance. Silero stays the always-on gate, the Redux head suits long recordings that are mostly speech, and the gate is opt-in. The fusion gain of the synthetic experiment did not hold on real recordings (+0.10 and +0.46 F1 points, no word error rate gain), so fusion is not built. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] --- AGENTS.md | 2 + README.md | 5 +- docs/vad-benchmarks.md | 285 +++++++++++++++++++++++++++++++++++++++-- docs/vad.md | 142 +++++++++++++++++++- 4 files changed, 417 insertions(+), 17 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index db8baf0..78dcbfe 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -158,6 +158,8 @@ tests/ ctest targets test_vad_head.cpp , VAD head probabilities (PARAKEET_TEST_GGUF_ULTRA) test_vad_segmenter.cpp , segmenter cut rules, 32 ms grid, Silero defaults, event tracker (model-independent) test_vad_options.cpp , VAD option parser, NULL and bad-file C-API paths (model-independent) + test_vad_run_gate.cpp , run gate (SegmenterOpts::run_gate) on synthetic probability streams (model-independent) + test_vad_run_gate_model.cpp, run gate on a clip with a noise stretch (PARAKEET_TEST_VAD_ONLY_REDUX_GGUF, _GGUF_ULTRA, _REDUX_KEEP, _REDUX_DEQ, PARAKEET_TEST_SILERO_GGUF) test_vad_only.cpp , VAD-only slice: C-API VAD equals the parent, other calls refuse it (PARAKEET_TEST_VAD_ONLY_GGUF, PARAKEET_TEST_GGUF_ULTRA) test_capi_vad_silero.cpp, Silero via the C-API: JSON, options, threads, stream (PARAKEET_TEST_SILERO_GGUF) test_transcribe_vad_silero.cpp, transcribe with Silero segments on a model without a head (PARAKEET_TEST_SILERO_GGUF, PARAKEET_TEST_GGUF) diff --git a/README.md b/README.md index 2114fa7..e13da27 100644 --- a/README.md +++ b/README.md @@ -182,8 +182,9 @@ parakeet-cli transcribe ... --stream # cache-aware st parakeet-cli transcribe ... --lang # Nemotron 3.5 language, default auto parakeet-cli transcribe ... --vad [--vad-model silero.gguf] # cut long audio at pauses (offline, greedy only) parakeet-cli transcribe ... --vad-trim SEC # trim each piece to its speech plus SEC (default 0.3, 0 = whole cuts) +parakeet-cli transcribe ... --vad-run-gate 0.92 # opt-in, for the Ultra/Redux head: drop speech runs with a low median probability (docs/vad.md) parakeet-cli transcribe ... --min-local-conf 0.5 # opt-in: drop words invented on noise (docs/vad.md) -parakeet-cli vad --model M --input A.wav [--mode segments] [--probabilities] # speech regions as JSON +parakeet-cli vad --model M --input A.wav [--mode segments] [--probabilities] [--run-gate P] # speech regions as JSON parakeet-cli scene --model ASR --diar DIAR --sound CED --input A.wav # words + speakers + sounds parakeet-cli scene ... --speakers SPK.gguf --registry R # name the speakers parakeet-cli enroll --model SPK.gguf --name Ada --input ada.wav --registry R @@ -319,7 +320,7 @@ VAD accuracy, speed and size, Silero against the Parakeet head against whisper.c - **Redux** runs on CPU only and offline only. SIMD kernels exist for x86-64 (AVX2, AVX-512 VNNI) and aarch64 with dotprod; other targets use a slow scalar kernel. See [ternary.md](docs/ternary.md). - **Ultra and Redux** are not NeMo-validated. Their parity is transcript-level against our own v3 path. -- **The VAD head** in Ultra and Redux gives false alarms on audio without speech. Use Silero as an always-on gate. See [vad.md](docs/vad.md). +- **The VAD head** in Ultra and Redux gives false alarms on audio without speech. Use Silero as an always-on gate. See [vad.md](docs/vad.md). The opt-in run gate (`--run-gate`, `--vad-run-gate`) removes most of the noise false alarms of the Redux head, not music. - **`transcribe --vad`** is offline and greedy decoding only. VAD-only files cannot transcribe. - **Bundles:** `bench` and the streaming ASR modes do not take a bundle. No bundle was run on a GPU. See [bundle.md](docs/bundle.md). - **Speaker identification** was measured on one two-voice fixture. See [speaker.md](docs/speaker.md). diff --git a/docs/vad-benchmarks.md b/docs/vad-benchmarks.md index 71da631..eea4381 100644 --- a/docs/vad-benchmarks.md +++ b/docs/vad-benchmarks.md @@ -768,8 +768,10 @@ In plain language: - The median-logit run gate is the best cheap option. It drops almost all false alarms and costs nothing for Redux on clean speech (95.4) and 0.5 F1 points for Ultra (3.3 on the noisy set). With the run threshold at 3.5 (not in the table) Redux clean F1 is 94.9 and - Ultra 93.6. It is not validated on WER and is not - implemented. On the 1168 s talk it removed 0.2 to 0.3 percent of the head's speech + Ultra 93.6. It is now implemented as the opt-in [run gate](vad.md#run-gate-opt-in) (a + median probability of 0.92 is a median logit of 2.5) and was measured on real recordings, see + [Real recordings](#real-recordings): it removes most noise false alarms of the Redux head, not music. + On the 1168 s talk it removed 0.2 to 0.3 percent of the head's speech frames at 2.5 (`res_E2_talk.txt`). In the inserted block test at most 18 percent of the block frames remained, and 0 in 21 of 24 files (`res_D_mit.md`). - The two-stage rule needs Silero. It removes the false alarms and keeps the F1 of the head. @@ -807,6 +809,12 @@ parakeet.cpp.** It is an offline experiment: the rules run in a Python port of t segmenter (`scripts/vad_bench/fusion/`), on probabilities saved from Silero and from the heads. No C++ code, API or option changed. +**Update: the gain did not hold on real recordings.** The +0.75 (Ultra) and +1.13 (Redux) F1 points +below come from synthetic clips. On 59 real recordings, against the best single detector tuned the same way, the +tuned fusion gained +0.10 points (Ultra) and +0.46 points (Redux), and it gave no word error rate gain. Fusion +was therefore not built. Keep this section as an experiment note and read +[Real recordings](#real-recordings) for the current conclusion. + ### Setup - Corpus: rebuilt from LibriSpeech test-clean. 342 speech clips (38 sets of 5 @@ -1078,6 +1086,250 @@ At 0.9 it costs real words, most on v3. Hence 0.5 is the suggested value. - The trim changes transcripts of long audio through the VAD paths slightly; the numbers above are from three talks and 12 clips per condition. +## Real recordings + +The earlier sections use synthetic clips and a few talks. This one repeats the main questions on real +recordings with human labels: is a head better than Silero, is fusing them worth it, what do the false +alarms cost, and does the choice of detector change the word error rate. It also measures the **run gate** +(see [vad.md](vad.md#run-gate-opt-in)), the one change that came out of it. + +Scripts, the recording list and the result tables are in +[`scripts/vad_bench/real_recordings/`](../scripts/vad_bench/real_recordings/README.md). + +In plain language: + +- On speech-dominant audio the choice of detector does not change the word error rate. All systems are + within 0.17 points on clean TED talks. +- The Redux head has the best default frame F1 (92.8) and loses the least speech. Silero is the most precise + and the only one that stays quiet on music, noise and environmental sounds. +- The heads call most of an hour of music or noise speech. The run gate removes most of the noise false alarms of + the Redux head and recovers about 80 percent of the word error rate lost on recordings with inserted music and + noise. It does not remove music. +- Fusing Silero and a head, the best idea of the synthetic study, gained 0.10 (Ultra) and 0.46 (Redux) F1 points + over the best single detector and nothing in word error rate. It was not built. + +### Setup + +| Part | Data | Size | +| --- | --- | --- | +| Speech | VoxConverse dev and test (36 recordings), AMI far-field `sdm` (9 meetings), AVA-Speech film clips with human labels (14 clips) | 12.1 h of audio, 9.5 h of labelled speech | +| No speech | MUSAN music (16 files), MUSAN noise (6), ESC-50 environmental sounds (5) | 2.6 h | +| Word error rate | Five TED-LIUM long-form talks, and three of them with non-speech clips inserted at pauses (the composite) | 0.8 h | + +- Splits are by recording. VoxConverse dev and AMI validation tune. VoxConverse test and AMI test are held out. + AVA and the non-speech files split by the parity of an MD5 of the id. Held out: 20 speech recordings, 3 of them + AMI meetings. +- Detectors: Silero F16, the Ultra head and the Redux head, from `parakeet-cli vad --probabilities`. Systems are + scored on a 10 ms grid. Frame precision, recall and F1 are pooled over the speech recordings. Intervals are 95 + percent bootstrap intervals over recordings. A Python replica of the segmenter was checked against the CLI. +- "Untuned" means no parameter was fitted: each detector with its own defaults. "Tuned" means chosen on the + tuning split and scored on the held-out split. +- The non-speech files have no speech by construction, so every frame called speech there is a false alarm. + +### Frame F1, untuned + +All recordings, percent. + +| System | VoxConverse | AMI | AVA | Pooled F1 | P | R | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| Silero, own defaults | 96.4 | 86.7 | 84.7 | 90.6 [88.4, 92.3] | 97.8 | 84.4 | +| Ultra head | 96.4 | 89.7 | 82.7 | 90.8 [89.3, 92.0] | 91.3 | 90.4 | +| Redux head | 97.3 | 92.2 | 85.4 | 92.8 [91.6, 93.7] | 92.8 | 92.7 | +| Fusion (two-stage rule), Ultra, defaults | 97.4 | 89.2 | 87.2 | 92.4 [90.5, 93.8] | 96.9 | 88.2 | +| Fusion (two-stage rule), Redux, defaults | 97.6 | 90.4 | 87.5 | 92.9 [91.1, 94.3] | 96.8 | 89.3 | +| Ultra head, run gate 0.92 | 96.4 | 85.4 | 85.4 | 90.2 [88.4, 91.7] | 96.5 | 84.7 | +| Redux head, run gate 0.92 | 96.9 | 90.8 | 86.2 | 92.5 [91.0, 93.7] | 97.0 | 88.3 | + +Reading it: Silero is precise (97.8) and misses speech (recall 84.4). The Redux head has the best F1 at its +defaults. The Ultra head is not clearly different from Silero (pooled difference 0.26 points, interval [-0.72, +1.36]). The heads have the higher recall and the lower precision. The gate moves a head toward Silero's +precision. Its pooled cost at 0.92 is 0.6 points for Ultra and 0.3 for Redux (paired differences -0.61 [-1.55, +0.39] and -0.32 [-0.98, 0.31], so neither is clearly different from zero). Per domain the gate costs 4.3 (Ultra) +and 1.4 (Redux) points on AMI, nothing on VoxConverse for Ultra and 0.4 for Redux, and gains 2.7 and 0.8 points +on AVA. + +### Tuned, held out + +Parameters (threshold, `min_speech`, `min_pause`, padding, and for the other rules their own) were chosen on the +tuning recordings and scored on the held-out ones. Pooled held-out F1, with the parameters chosen: + +| System | Held-out F1 [95% CI] | Chosen | +| --- | ---: | --- | +| Silero, tuned | 93.4 [91.2, 94.9] | threshold 0.1, `min_speech` 0.25, `min_pause` 0.5, pad 0.1 | +| Ultra head, tuned | 91.5 [89.4, 93.2] | threshold 0.8, `min_speech` 0.25, `min_pause` 0.5, pad 0.2 | +| Redux head, tuned | 93.3 [91.5, 94.7] | threshold 0.6, `min_speech` 0.5, `min_pause` 0.5, pad 0.1 | +| Fusion, Ultra, tuned | 93.5 [91.4, 95.0] | | +| Fusion, Redux, tuned | 93.8 [91.8, 95.3] | | +| Redux head with run gate 0.8, tuned | 93.9 [92.2, 95.2] | threshold 0.5 (gate 0.8), `min_speech` 0.25, `min_pause` 0.5, pad 0.03 | + +Untuned on the same held-out recordings the scores are 89.1 (Silero), 89.3 (Ultra) and 91.4 (Redux), so tuning +adds 4.3, 2.2 and 1.9 points. Paired against the best single detector tuned the same way (Silero, for both), the +tuned fusion gains **0.10 points [-0.03, 0.26] with Ultra and 0.46 points [0.30, 0.66] with Redux**. That is less +than the +0.75 and +1.13 of the [synthetic experiment](#fusing-silero-and-the-head-offline-experiment). By +domain (VoxConverse / AMI / AVA) the gain is -0.07 / -0.29 / +0.58 with Ultra and +0.15 / +0.65 / +0.71 with +Redux. + +### Fusion: not worth a flag + +The fusion rule needs both detectors to run. It gains 0.10 points (Ultra) and 0.46 points (Redux) over the best +tuned single detector, and it gave **no word error rate gain** (see below). **Fusion is not implemented and there is +no flag for it.** The synthetic section above stays as an experiment note. + +### The reference is noisy + +Part of what tuning gains comes from how the references are written, not from the detectors: + +- In VoxConverse and AMI the frames Silero "misses" are mostly pauses inside annotated turns. The annotators mark + a whole turn and the metric counts the pauses in it as speech. Tuning therefore pushes `min_pause` to 0.5 s + (bridge the pauses) and the padding to 0.1 s. That says something about the labelling convention and little + about the detector. +- With a collar of 0.1 s around every reference boundary (cells within 0.1 s of a change are ignored), every F1 + rises by about 0.7 to 0.9 points and the ranking does not change: + +| System | F1, no collar | F1, collar 0.1 s | F1, collar 0.25 s | +| --- | ---: | ---: | ---: | +| Silero, own defaults | 90.6 | 91.3 | 91.7 | +| Ultra head | 90.8 | 91.5 | 91.9 | +| Redux head | 92.8 | 93.5 | 93.9 | +| Fusion, Ultra, defaults | 92.4 | 93.1 | 93.6 | +| Fusion, Redux, defaults | 92.9 | 93.7 | 94.2 | +| Redux head, run gate 0.92 | 92.5 | 93.2 | 93.8 | + +Read an F1 difference of a point or less as within the reference noise. + +### False alarms on audio without speech + +Seconds called speech per hour of audio, with 95 percent intervals over files where given. Music by MUSAN source +is in `results/extras.md`. + +| System | Music | Noise | ESC-50 | Pooled s/h | Pooled regions/h | +| --- | ---: | ---: | ---: | ---: | ---: | +| Silero, own defaults | 135 | 1 | 1 | 48 [0, 136] | 27 | +| Ultra head | 2823 | 1979 | 1605 | 2174 [1882, 2524] | 794 | +| Ultra head, run gate 0.92 | 1625 | 398 | 395 | 826 [523, 1229] | 158 | +| Redux head | 2123 | 1599 | 1236 | 1685 [1435, 1987] | 1013 | +| Redux head, run gate 0.92 | 644 | 23 | 131 | 269 [122, 482] | 85 | +| Fusion, Ultra, defaults | 165 | 2 | 5 | 60 | 24 | +| Fusion, Redux, defaults | 164 | 2 | 5 | 59 | 23 | + +The Silero music figure comes from one MUSAN source (Jamendo, 294 s/h); the other four sources give 0 to 5. + +The run gate (keep a run of `p >= 0.5` only when the median of its frames is at least the gate) trades speech +for false alarms along one curve. F1 is the pooled F1 of the head at threshold 0.5 with bridge 0.1 s, +`min_speech` 0.1 s, `min_pause` 0.2 s and no padding: + +| Head | Gate | Pooled F1 | Music s/h | Noise s/h | ESC-50 s/h | +| --- | ---: | ---: | ---: | ---: | ---: | +| Redux | off | 92.8 | 2123 | 1599 | 1236 | +| Redux | 0.8 | 93.5 | 1305 | 639 | 377 | +| Redux | 0.9 | 92.8 | 762 | 79 | 167 | +| Redux | 0.92 | 92.5 | 644 | 23 | 131 | +| Redux | 0.96 | 91.0 | 265 | 4 | 34 | +| Redux | 0.98 | 88.6 | 107 | 1 | 15 | +| Ultra | off | 90.8 | 2823 | 1979 | 1605 | +| Ultra | 0.8 | 91.2 | 2389 | 1226 | 959 | +| Ultra | 0.9 | 90.7 | 1831 | 576 | 488 | +| Ultra | 0.92 | 90.2 | 1625 | 398 | 395 | +| Ultra | 0.96 | 87.7 | 761 | 31 | 178 | +| Ultra | 0.98 | 83.4 | 339 | 9 | 80 | + +Redux at 0.98 reaches 107 / 1 / 15 s/h but costs 4.2 F1 points. Ultra at 0.98 costs 7.4 points and music is +still at 339 s/h. The gate is not a music rejector: music runs have a high median like speech. A lower gate +costs less speech and removes less: Redux at 0.8 gains 0.7 F1 points and removes 39 to 70 percent of the false +alarms, Ultra at 0.8 removes 15 to 40 percent. + +### What the decoder receives + +`transcribe --vad` hands the decoder the segments from `segment_by_vad` with the trim at 0.3 s (the default). +Seconds of reference non-speech inside the decoded segments, per hour of audio: + +| System | VoxConverse | AMI | AVA | +| --- | ---: | ---: | ---: | +| Silero | 273 | 355 | 885 | +| Ultra head | 317 | 598 | 1348 | +| Redux head | 296 | 620 | 1328 | +| Ultra head, run gate 0.92 | 258 | 412 | 927 | +| Redux head, run gate 0.92 | 238 | 426 | 832 | +| Oracle (reference speech mask) | 230 | 375 | 833 | + +Speech lost (labelled speech outside the segments, percent, VoxConverse / AMI / AVA): Silero 0.2 / 5.3 / 6.3, +Ultra head 0.0 / 0.6 / 0.3, Redux head 0.1 / 0.3 / 0.2, Ultra head with the gate 0.9 / 3.9 / 5.1, Redux head with +the gate 0.9 / 2.1 / 7.8. The heads lose the least speech. + +Of one hour of non-speech audio, the share the decoder receives (trim 0.3), in percent: + +| System | Music | Noise | ESC-50 | +| --- | ---: | ---: | ---: | +| Silero | 6 | 0.03 | 0.03 | +| Ultra head | 95 | 84 | 94 | +| Redux head | 91 | 82 | 94 | +| Ultra head, run gate 0.92 | 53 | 14 | 27 | +| Redux head, run gate 0.92 | 27 | 1.1 | 5.6 | + +A head hands the decoder most of an hour of music or noise. The gate cuts that share by half or more for Redux, +and it stays far above Silero. + +### Word error rate + +The ASR models are the ones of the heads (Ultra Q8_0 and the packed Redux), decoding the segments each system +gives. + +**Clean TED talks** (5 talks, about 2880 s decoded). The detector does not matter on speech-dominant audio. WER in +percent: + +| ASR model | Head | Silero | Fusion, defaults | Fusion, tuned | Run gate 0.92 | +| --- | ---: | ---: | ---: | ---: | ---: | +| Ultra | 4.07 | 4.18 | 4.18 | 4.21 | 4.24 | +| Redux | 5.02 | 5.02 | 5.02 | 4.97 | 4.97 | + +All systems are within 0.17 points (Ultra) and 0.05 points (Redux) of each other, and every paired interval +includes zero. + +**Composite recordings.** Three of the talks with 40 s of music, 30 s of noise and 40 s of vocal music inserted at +pauses. The reference is the talk, so every word decoded over the inserted audio is an insertion. It is synthetic +by construction and has three talks. WER in percent: + +| ASR model | Head | Silero | Fusion, defaults | Run gate 0.92 | +| --- | ---: | ---: | ---: | ---: | +| Ultra | 5.91 [3.66, 8.75] | 3.87 [2.91, 4.90] | 4.16 | 4.24 [2.97, 5.79] | +| Redux | 6.94 [4.76, 9.55] | 4.76 [3.85, 5.71] | 5.02 | 5.22 [4.25, 6.24] | + +The gate recovers about 80 percent of the head's penalty against Silero (Ultra: 1.67 of 2.04 points; Redux: 1.72 +of 2.18). Silero is still the best on this audio. The decoded seconds are 1137 (Silero), 1377 and 1367 (Ultra and +Redux head) and 1160 and 1143 (gate). + +### The 30 s hard cut + +The segmenter cuts at the last pause of a 30 s window, and when there is none it cuts at 30 s inside speech. The +heads hard-cut more often than Silero on VoxConverse: 9.7 hard cuts per hour for Silero, 32.0 for Ultra and 35.1 +for Redux (0.5 to 7.7 per hour on AMI and AVA). + +The measured cost of a forced cut is small. On the five TED talks, cutting Silero's speech mask every 20 s +regardless of pauses (142 hard cuts in 0.8 h, against 2 for cutting at pauses) adds 26 word errors with the Ultra +model and 28 with the Redux model, which is **about 0.18 to 0.2 word errors per forced cut**. At 35 cuts per hour +that is about 7 word errors per hour, or 0.06 points of WER on talks of about 11 000 words per hour. After the +trim, the hard cut is not the main cost. + +What the decoder still gets that is not speech is inside the pieces. Edge margins are 44 to 75 s/h. The rest are +pauses that stay inside a segment: for the Redux head, pauses under 1 s, from 1 to 5 s and of 5 s or more add up to +140 / 86 / 35 s/h on VoxConverse, 297 / 358 / 102 on AMI and 370 / 535 / 172 on AVA. Cutting at every pause of 2 s +or more (instead of only at the last pause of a window) lowers the decoded non-speech by 22 to 59 percent but +loses more speech: with the Redux head, speech lost goes from 0.1 / 0.3 / 0.2 to 0.1 / 0.8 / 1.3 percent, and with +Silero from 0.2 / 5.3 / 6.3 to 0.5 / 9.0 / 11.1 (VoxConverse / AMI / AVA). Lost speech is a worse error than a +decoded pause, so the cut rule is unchanged. + +### Limits + +- DIHARD, CallHome and the MUSAN speech files were not used: their speech labels were not usable for frame scoring. +- WER was measured on TED talks only. The meetings and film clips have no word-level reference here. +- The composite recordings are synthetic (three talks, clips inserted at pauses). They show the effect of long + non-speech stretches, not an average over real audio. +- The cut policies were measured by non-speech seconds and lost speech, not by WER. +- The Silero ONNX model was not run in this study. The Silero numbers are from the parakeet.cpp GGUF. +- Held out: 20 recordings, and only 3 AMI meetings, so the held-out intervals are wide. +- The reference noise described above applies to VoxConverse and AMI. Differences of about a point between + systems are within it. + ## When to use which This is limited to what the numbers above support. @@ -1088,7 +1340,21 @@ This is limited to what the numbers above support. synthetic data and 91.8 to 92.1 on the TED talks. - **Always-on gate, or audio with long stretches without speech.** Silero. The heads give false alarms on noise-only audio, and the effect depends on the noise type and level (see - [the head on noise-only audio](#the-head-on-noise-only-audio)); Silero gave none. + [the head on noise-only audio](#the-head-on-noise-only-audio)); Silero gave none. On real + recordings Silero called 48 s of an hour of music, noise and environmental sounds speech, the Ultra head 2174 s and + the Redux head 1685 s. Use Silero with its own defaults. Lowering its threshold to 0.2 or 0.3 raises the pooled F1 on + real speech from 90.6 to 92.4 or 91.8 and keeps music at or below 199 s/h (noise and ESC-50 at 5 s/h or less); see + [Real recordings](#real-recordings). +- **The Redux head, for long recordings that are mostly speech** (talks, meetings, interviews). On 59 real + recordings it had the best default F1 (92.8 against 90.6 for Silero and 90.8 for the Ultra head), lost the least + speech (0.1, 0.3 and 0.2 percent on VoxConverse, AMI and AVA, against 0.2, 5.3 and 6.3 for Silero), and gave the same + word error rate as Silero on clean TED talks. +- **Recordings with long stretches of music or noise.** Silero. Behind a head the decoder receives most of an + hour of music or noise, and the word error rate on the composite recordings was 6.94 percent (Redux head) against 4.76 (Silero). +- **The run gate, opt-in, for a head.** It drops a speech run whose median probability is below the gate. At 0.92 it + cut the Redux head's false alarms from 1685 to 269 s/h and recovered about 80 percent of the word error rate lost + on the composite recordings, at a pooled cost of 0.3 F1 points (0.6 for Ultra, which gains less). It does not + remove music and it is not a noise rejector. See [vad.md](vad.md#run-gate-opt-in). - **The Parakeet head when Ultra or Redux is already loaded, on audio that is mostly speech** (for example recorded talks before transcription). The head costs no extra model and its scores are close to Silero in the same tests: within about 1.5 F1 points @@ -1109,12 +1375,14 @@ This is limited to what the numbers above support. probabilities of parakeet.cpp match onnxruntime more closely (mean abs diff 0.00002, 2 flipped frames) than those of whisper.cpp do (0.00224, 309 flipped frames), but this did not change the segment scores. -- **Both together.** In an offline experiment, Silero deciding and the head only moving +- **Both together.** In an offline experiment on synthetic clips, Silero deciding and the head only moving the edges (the two-stage rule) gave 0.75 to 1.13 more F1 points than the best single - detector and no false alarms on noise. It is not implemented in parakeet.cpp. See - [the fusion experiment](#fusing-silero-and-the-head-offline-experiment). -- **Not supported by these numbers:** any claim about GPUs, ARM, other languages, - music or non speech noise beyond the synthetic signals of [the noise study](#root-cause-study), streaming latency, or the effect of the VAD on WER. + detector and no false alarms on noise. On real recordings the gain was 0.10 (Ultra) and 0.46 (Redux) + points and there was no word error rate gain, so it is not implemented in parakeet.cpp. See + [the fusion experiment](#fusing-silero-and-the-head-offline-experiment) and + [Real recordings](#real-recordings). +- **Not supported by these numbers:** any claim about GPUs, ARM, other languages, streaming latency, or the effect of + the VAD on WER beyond the TED talks and the composite recordings of [Real recordings](#real-recordings). ## How to reproduce @@ -1131,6 +1399,7 @@ are committed. In short: 5. Long talks: `longform_b1.sh`. 6. Noise root-cause study: [`noise_dive/`](../scripts/vad_bench/noise_dive/README.md). 7. Segment trim and word filter: [`decoder_guards/`](../scripts/vad_bench/decoder_guards/README.md). +8. Real recordings and the run gate: [`real_recordings/`](../scripts/vad_bench/real_recordings/README.md). Small result files of the runs on this page (tables, per run timings and load logs) are in `scripts/vad_bench/results/`. The raw per clip predictions are not committed; diff --git a/docs/vad.md b/docs/vad.md index 3fb7a1a..6e3d1b7 100644 --- a/docs/vad.md +++ b/docs/vad.md @@ -47,13 +47,41 @@ that do and do not trigger it, and the mitigations we tested are in For `transcribe --vad` the user-visible cost is small: some wasted decoding, and rarely a hallucinated phrase. The 30 s hard cut of the segmenter can also keep a large part of a long noise gap, even where the head called none of it speech. This is a segmenter -behaviour, not a head false alarm, and it has not been fixed. +behaviour, not a head false alarm. On real recordings it is not the main cost after the +trim (see [the benchmark page](vad-benchmarks.md#the-30-s-hard-cut)). + +**On real recordings** (36 VoxConverse recordings, 9 AMI far-field meetings and 14 AVA film +clips, 12.1 h with 9.5 h of labelled speech, and 2.6 h of music, noise and ESC-50 clips) the +picture is the same, with numbers: + +- Frame F1 on speech, at the defaults: Silero 90.6, Ultra head 90.8, Redux head 92.8. The + Redux head has the best default score and loses the least speech. +- On audio without speech, seconds called speech per hour: Silero 48, Ultra head 2174, + Redux head 1685. The head calls most of a music or noise hour speech. +- On clean TED talks the choice of detector does not change the word error rate: all systems + are within 0.17 percentage points. With 40 s of music, 30 s of noise and 40 s of vocal music + inserted into each talk, the word error rate is 3.87 percent (Ultra) and 4.76 percent (Redux) + when Silero cuts the audio, and 5.91 and 6.94 percent when the head does. + +So the advice is: + +- **Silero with its own defaults is the always-on gate.** Lowering its `threshold` to 0.2 to + 0.3 gives more recall (pooled F1 92.4 and 91.8 against 90.6) and at most 199 s of false + alarm per hour on music (noise and ESC-50: 5 s/h or less). +- **The Redux head for long recordings that are mostly speech** (talks, meetings, interviews): + the best default F1, the least speech lost, and the same word error rate as Silero. +- **Silero for recordings with long stretches of music or noise.** +- **The [run gate](#run-gate-opt-in) is an opt-in option for the heads** (0.92 to 0.96 for + Redux; Ultra gains less). It removes most of the noise false alarms. It does not remove music + and it is not a noise rejector. An offline experiment also combined the two: Silero decides what is speech, and the head -only moves the edges. It scored 0.75 to 1.13 F1 points above the best single detector on -the synthetic clips and gave no false alarms on noise. It is not implemented in -parakeet.cpp. See -[Fusing Silero and the head](vad-benchmarks.md#fusing-silero-and-the-head-offline-experiment). +only moves the edges. On synthetic clips it scored 0.75 to 1.13 F1 points above the best +single detector. That did not hold on real recordings: against the best single detector, tuned +the same way, the gain was 0.10 points for Ultra and 0.46 for Redux, with no change in word error +rate. It is not implemented in parakeet.cpp and there are no plans to add it. See +[Fusing Silero and the head](vad-benchmarks.md#fusing-silero-and-the-head-offline-experiment) +and [Real recordings](vad-benchmarks.md#real-recordings). ## Standalone VAD API @@ -68,7 +96,7 @@ char* parakeet_capi_vad_path_json(parakeet_ctx* ctx, const char* wav_path, const parakeet-cli vad --model --input audio.wav \ [--mode speech|segments] [--probabilities] [--threshold F] [--min-pause SEC] \ - [--min-speech SEC] [--speech-pad SEC] [--max-segment SEC] [--threads N] + [--min-speech SEC] [--speech-pad SEC] [--max-segment SEC] [--trim SEC] [--run-gate P] [--threads N] ``` Both return NULL on error with the message in `parakeet_capi_last_error`, and the @@ -99,6 +127,7 @@ the same keys). Unknown keys and out of range values are errors. | `speech_pad` | seconds >= 0, `speech` mode: widen each region on both sides | 0 | 0.03 | | `max_segment` | seconds; cap in `segments` mode | 30 | 30 | | `trim` | seconds >= 0; `segments` mode and the transcribe functions: shrink each cut to its speech plus this much on each side, 0 = keep the whole cut | 0.3 | 0.3 | +| `run_gate` | probability in [0, 1); drop a speech run whose median frame probability is below it, 0 = off (see [Run gate](#run-gate-opt-in)) | 0 | 0 | | `mode` | `speech` or `segments` | `speech` | `speech` | | `probabilities` | add the per frame probabilities | false | false | @@ -132,6 +161,100 @@ long noisy stretches the decoder sees much less noise, and with Silero the trim stops whole sentences from being dropped on clean speech. Numbers: [vad-benchmarks.md](vad-benchmarks.md#trimming-segments-and-the-word-filter). +## Run gate (opt-in) + +The head calls steady noise and music speech, and the gate is a cheap check on that. A speech +run is a stretch of consecutive frames with `p >= threshold`. With `run_gate` set, a run is +dropped when the **median** of the probabilities of its frames is below the gate. Speech frames +sit at a logit of +5 to +12 (probability 0.99 and above). A steady noise run sits on a plateau +at a logit of +1.5 to +2 (0.8 to 0.9), with some higher peaks, so its median is lower than its +peak. Off by default (`run_gate` 0): +the output is then byte for byte what it was before the option existed. + +``` +parakeet-cli vad --model redux-vad.gguf --input a.wav --run-gate 0.92 [--mode segments] +parakeet-cli transcribe --model redux.gguf --input long.wav --vad --vad-run-gate 0.92 +{"run_gate":0.92} # in the options JSON of the vad and transcribe C functions +``` + +Definition (the same for every detector and both frame sizes): + +- A run is a maximal stretch of frames with `p >= threshold`, found before any bridging. A gap that + the segmenter later bridges is never inside a run, so a run does not borrow the probability of its + neighbour. +- The median is over the probabilities of the frames of the run. For an even number of frames it is + the mean of the two middle values. A run of one frame has its own probability as median. +- The run is kept when `median >= run_gate` and dropped when it is below. A median equal to the gate + keeps the run. +- The gate acts first. The frames of a dropped run count as silence for bridging, `min_speech`, the + pauses, the cuts and the trim. It applies to both modes (`speech` and `segments`), to every + detector, and to the transcribe functions that take the VAD options. +- The value must be a number in [0, 1). Other values, and non-numbers, are errors, like every + other key. + +The gate is **offline only**. The streaming event tracker (`parakeet_capi_vad_stream_*`) decides frame +by frame and has no run median, so `parakeet_capi_vad_stream_begin` refuses a non-zero `run_gate`. +Audio of at most `max_segment` seconds is returned whole in `segments` mode without running the +segmenter, so the gate has no effect there (as for the trim). + +Which value: **0.92 to 0.96 for the Redux head.** For Ultra the gate is weaker (see below). Silero +accepts the option, but its noise runs are rare, and the gate is meant for the heads. + +What it does on real recordings (the full study is in +[vad-benchmarks.md](vad-benchmarks.md#real-recordings)). Frame F1 on speech, untuned, in percent: + +| System | VoxConverse | AMI | AVA | Pooled | +| --- | ---: | ---: | ---: | ---: | +| Ultra head | 96.4 | 89.7 | 82.7 | 90.8 | +| Ultra head, gate 0.92 | 96.4 | 85.4 | 85.4 | 90.2 | +| Redux head | 97.3 | 92.2 | 85.4 | 92.8 | +| Redux head, gate 0.92 | 96.9 | 90.8 | 86.2 | 92.5 | +| Silero, own defaults | 96.4 | 86.7 | 84.7 | 90.6 | + +The pooled cost of the gate at 0.92 is 0.6 points for Ultra and 0.3 for Redux. Both intervals +include zero. Seconds called speech per hour of audio without speech: + +| System | Music | Noise | ESC-50 | Pooled | +| --- | ---: | ---: | ---: | ---: | +| Silero | 135 | 1 | 1 | 48 | +| Ultra head | 2823 | 1979 | 1605 | 2174 | +| Ultra head, gate 0.92 | 1625 | 398 | 395 | 826 | +| Redux head | 2123 | 1599 | 1236 | 1685 | +| Redux head, gate 0.92 | 644 | 23 | 131 | 269 | + +The gate trades recall for false alarms. For the Redux head at 0.98 the music figure falls to 107 +s/h, noise to 1 and ESC-50 to 15, but F1 on speech falls by 4.2 points. For the Ultra head at 0.98 +the cost is 7.4 points and music is still 339 s/h. In `transcribe --vad` the effect is on the +decoded audio. Of one hour of non-speech with the trim at 0.3, the decoder gets this share: + +| System (percent of the hour) | Music | Noise | ESC-50 | +| --- | ---: | ---: | ---: | +| Silero | 6 | 0.03 | 0.03 | +| Ultra head | 95 | 84 | 94 | +| Ultra head, gate 0.92 | 53 | 14 | 27 | +| Redux head | 91 | 82 | 94 | +| Redux head, gate 0.92 | 27 | 1.1 | 5.6 | + +Word error rate: on clean TED talks every system is within 0.17 points, gate or not. On the three +talks with 40 s of music, 30 s of noise and 40 s of vocal music inserted, the gate recovers about +80 percent of the head's penalty against Silero: Ultra 5.91 (head), 4.24 (gate), 3.87 (Silero); +Redux 6.94, 5.22, 4.76. + +What the gate does **not** do: + +- **Music still triggers the head.** At 0.92 the Redux head still calls 644 s of an hour of music + speech, and the Ultra head 1625 s. Music has a high median like speech does. +- **It is not a noise rejector.** It lowers the false alarms of the head on noise, but some remain: + with the Redux head 23 s/h of noise and 131 s/h of ESC-50, with the Ultra head 398 and 395 s/h, + against about 1 s/h for Silero. Use Silero where false alarms cost something. +- **It costs speech.** On AMI the F1 falls by 1.4 (Redux) and 4.3 (Ultra) points, and in `segments` + mode on AVA the Redux gate loses 7.8 percent of the labelled speech (the head alone loses 0.2). + It raises F1 on AVA, where the head over-detects, so the pooled cost is small. + +Cross-check: the C++ gate gives the same regions as the Python gate of the study on 688 +combinations of recording, detector and gate value, and the same speech seconds removed +(`scripts/vad_bench/real_recordings/verify_gate.py`). + ## Word filter (opt-in) A confidence filter can remove the words a model invents on noise. It is @@ -161,7 +284,7 @@ char* parakeet_capi_transcribe_path_json_with(parakeet_ctx* ctx, const char* wav The options JSON of `parakeet_capi_transcribe_path_json_with` takes the three keys above. `parakeet_capi_transcribe_path_json_vad_with` takes them too, next to -the VAD keys (`trim` included). With a filter on, the JSON document gets one more +the VAD keys (`trim` and `run_gate` included). With a filter on, the JSON document gets one more member, `"guard":{"dropped_words":N}` (N is 0 when nothing was dropped), and the dropped words are also removed from `text`, `words` and `tokens`. @@ -359,6 +482,11 @@ machine, not a benchmark. For benchmarks, see [vad-benchmarks.md](vad-benchmarks - `test_vad_segmenter` (no model) covers the 32 ms grid, the Silero defaults, the padding and the streaming event tracker. `test_vad_options` (no model) covers the option parser and the NULL and bad file paths of the C-API. +- `test_vad_run_gate` (no model) covers the run gate on synthetic probability streams at 80 ms and + 32 ms frames: runs above and below the gate, a low median with a high peak, bridged gaps, a + one frame run, the boundary (a median equal to the gate keeps the run), the trim, and that gate 0 does not change + the output. `test_vad_run_gate_model` (label `model`, the VAD-only slice, Ultra, Redux and Silero + files) runs the gate on a clip with a noise stretch. - `test_capi_vad_silero` (label `model`, `PARAKEET_TEST_SILERO_GGUF`) covers the C-API document at both rates, options, errors, threads and the stream. - `test_transcribe_vad_silero` (label `model`, `PARAKEET_TEST_SILERO_GGUF` and From a9daf795ef981bb0c84c410d02557f9ee0c8f6b8 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Tue, 6 Oct 2026 07:39:52 +0000 Subject: [PATCH 3/3] scripts: add the real-recording VAD study Add the scripts of the study (data fetch, probability dump, a Python replica of the segmenter, frame and decoder scoring, WER reports) and the small result tables, with a README on how to rerun them. Paths are environment variables. verify_gate.py checks the C++ run gate against the Python gate on stored probabilities. No audio, probabilities or models. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] --- scripts/vad_bench/real_recordings/README.md | 79 ++ .../vad_bench/real_recordings/dump_probs.py | 28 + .../vad_bench/real_recordings/eval_frames.py | 53 ++ scripts/vad_bench/real_recordings/extras.py | 75 ++ .../vad_bench/real_recordings/fetch_ava.py | 17 + .../vad_bench/real_recordings/fetch_diar.py | 21 + .../real_recordings/fetch_nonspeech.py | 35 + .../real_recordings/gate_segtool.cpp | 15 + .../vad_bench/real_recordings/mk_composite.py | 31 + .../vad_bench/real_recordings/ref_check.py | 17 + .../real_recordings/report_frames.py | 177 +++++ .../real_recordings/report_fusion_grid.py | 16 + .../real_recordings/report_options.py | 29 + .../real_recordings/results/extras.md | 37 + .../real_recordings/results/frames.md | 244 +++++++ .../real_recordings/results/fusion_grid.md | 26 + .../real_recordings/results/options.md | 43 ++ .../real_recordings/results/recordings.json | 690 ++++++++++++++++++ .../vad_bench/real_recordings/results/seg.md | 143 ++++ .../real_recordings/results/tuned.json | 1 + .../vad_bench/real_recordings/results/wer.md | 25 + .../real_recordings/results/wer_comp.md | 25 + .../real_recordings/results/wer_forced.md | 17 + scripts/vad_bench/real_recordings/rl.py | 44 ++ scripts/vad_bench/real_recordings/seg_eval.py | 62 ++ .../vad_bench/real_recordings/seg_report.py | 49 ++ scripts/vad_bench/real_recordings/systems.py | 55 ++ .../vad_bench/real_recordings/verify_gate.py | 62 ++ .../real_recordings/verify_native.py | 13 + .../vad_bench/real_recordings/verify_seg.py | 20 + scripts/vad_bench/real_recordings/vr.py | 371 ++++++++++ scripts/vad_bench/real_recordings/wer_dec.py | 40 + scripts/vad_bench/real_recordings/wer_plan.py | 43 ++ .../real_recordings/wer_plan_forced.py | 14 + .../vad_bench/real_recordings/wer_report.py | 60 ++ 35 files changed, 2677 insertions(+) create mode 100644 scripts/vad_bench/real_recordings/README.md create mode 100644 scripts/vad_bench/real_recordings/dump_probs.py create mode 100644 scripts/vad_bench/real_recordings/eval_frames.py create mode 100644 scripts/vad_bench/real_recordings/extras.py create mode 100644 scripts/vad_bench/real_recordings/fetch_ava.py create mode 100644 scripts/vad_bench/real_recordings/fetch_diar.py create mode 100644 scripts/vad_bench/real_recordings/fetch_nonspeech.py create mode 100644 scripts/vad_bench/real_recordings/gate_segtool.cpp create mode 100644 scripts/vad_bench/real_recordings/mk_composite.py create mode 100644 scripts/vad_bench/real_recordings/ref_check.py create mode 100644 scripts/vad_bench/real_recordings/report_frames.py create mode 100644 scripts/vad_bench/real_recordings/report_fusion_grid.py create mode 100644 scripts/vad_bench/real_recordings/report_options.py create mode 100644 scripts/vad_bench/real_recordings/results/extras.md create mode 100644 scripts/vad_bench/real_recordings/results/frames.md create mode 100644 scripts/vad_bench/real_recordings/results/fusion_grid.md create mode 100644 scripts/vad_bench/real_recordings/results/options.md create mode 100644 scripts/vad_bench/real_recordings/results/recordings.json create mode 100644 scripts/vad_bench/real_recordings/results/seg.md create mode 100644 scripts/vad_bench/real_recordings/results/tuned.json create mode 100644 scripts/vad_bench/real_recordings/results/wer.md create mode 100644 scripts/vad_bench/real_recordings/results/wer_comp.md create mode 100644 scripts/vad_bench/real_recordings/results/wer_forced.md create mode 100644 scripts/vad_bench/real_recordings/rl.py create mode 100644 scripts/vad_bench/real_recordings/seg_eval.py create mode 100644 scripts/vad_bench/real_recordings/seg_report.py create mode 100644 scripts/vad_bench/real_recordings/systems.py create mode 100644 scripts/vad_bench/real_recordings/verify_gate.py create mode 100644 scripts/vad_bench/real_recordings/verify_native.py create mode 100644 scripts/vad_bench/real_recordings/verify_seg.py create mode 100644 scripts/vad_bench/real_recordings/vr.py create mode 100644 scripts/vad_bench/real_recordings/wer_dec.py create mode 100644 scripts/vad_bench/real_recordings/wer_plan.py create mode 100644 scripts/vad_bench/real_recordings/wer_plan_forced.py create mode 100644 scripts/vad_bench/real_recordings/wer_report.py diff --git a/scripts/vad_bench/real_recordings/README.md b/scripts/vad_bench/real_recordings/README.md new file mode 100644 index 0000000..71b5ebf --- /dev/null +++ b/scripts/vad_bench/real_recordings/README.md @@ -0,0 +1,79 @@ +# Real-recording VAD study + +Scripts and result tables behind the section +[Real recordings](../../../docs/vad-benchmarks.md#real-recordings) of +`docs/vad-benchmarks.md` and the run gate of [docs/vad.md](../../../docs/vad.md#run-gate-opt-in). +No audio, probability file or model is committed, only the scripts and the small result tables +in `results/` (about 100 kB). + +The study compares Silero, the Ultra head and the Redux head on real recordings: +59 recordings with human speech labels (12.1 h of audio, 9.5 h of labelled speech) and 27 files +without speech (2.6 h). `results/recordings.json` lists every recording with its domain, split, +duration and labelled speech seconds. + +| Domain | Source | Recordings | Audio | +| --- | --- | ---: | ---: | +| `vox` | VoxConverse (`diarizers-community/voxconverse`), dev and test splits, every 8th recording | 36 | 4.4 h | +| `ami` | AMI far-field (`diarizers-community/ami`, config `sdm`), validation every 3rd and test every 2nd | 9 | 4.2 h | +| `ava` | AVA-Speech film clips with human labels (`nccratliri/vad-human-ava-speech`), chosen by an MD5 order | 14 | 3.5 h | +| `music` | MUSAN music (`corypaik/musan`, config `music`) | 16 | 0.9 h | +| `noise` | MUSAN noise (`corypaik/musan`, config `noise`) | 6 | 1.0 h | +| `esc` | ESC-50 (`ashraq/esc50`), every 4th clip, joined into long files | 5 | 0.7 h | + +Tuning and test are disjoint by recording: VoxConverse dev and AMI validation tune, VoxConverse test and +AMI test are held out, AVA and the non-speech files split by the parity of an MD5 of the id. Only 20 +speech recordings are held out (AMI: 3 meetings), so the held-out intervals are wide. +The TED-LIUM long-form talks of the WER part come from `../fetch_ted_talks.py`. + +## Setup + +All paths are arguments or environment variables. Run from any directory. + +| Variable | Meaning | Default | +| --- | --- | --- | +| `VAD_REAL_ROOT` | work directory: `data//` (audio and labels), `probs/`, `results/` | current directory | +| `PARAKEET_CLI` | a `parakeet-cli` built from master (any build with `vad --probabilities`) | `$VAD_REAL_ROOT/parakeet-cli-clean` | +| `VAD_REAL_MODELS` | directory with `silero-vad-f16.gguf`, `ultra-vad-q8_0.gguf`, `redux-vad.gguf` (the VAD-only slices; for `wer_dec.py` also `ultra-q8_0.gguf` and `redux-packed.gguf`) | `$VAD_REAL_ROOT/models` | +| `HF_HOME` | Hugging Face cache for the fetch scripts | `$VAD_REAL_ROOT/hfhome` | + +Python packages: `numpy numba soundfile librosa datasets huggingface_hub`; the fetch of AVA also needs `ffmpeg`. + +## Steps + +1. Fetch the data. `fetch_diar.py NAME CONFIG SPLITS K CAP_HOURS OUT_DIR` keeps the recordings whose index + modulo K equals K//2 in each split, up to the hour cap and writes a 16 kHz mono WAV and a JSON + with the speaker turns. `fetch_ava.py OUT_DIR N` downloads N clips. `fetch_nonspeech.py music|noise|esc OUT_DIR ...` + writes the non-speech files. The exact recordings of the published run are in `results/recordings.json`. +2. `dump_probs.py` runs `parakeet-cli vad --probabilities` for every recording and detector and stores the + per-frame probabilities (`probs/*.npy`) next to the CLI's own speech regions. +3. `verify_native.py` checks that the Python segmenter replica in `vr.py` gives the CLI's regions; + `verify_seg.py` does the same for the `segments` mode. +4. `eval_frames.py` scores every system on a 10 ms grid (frame precision, recall, F1) and `report_frames.py`, + `report_options.py`, `report_fusion_grid.py` and `extras.py` print the tables of `results/frames.md`, + `options.md`, `fusion_grid.md` and `extras.md` (confidence intervals are bootstrap over recordings). +5. `seg_eval.py` and `seg_report.py` measure what the segmenter sends to the decoder (`results/seg.md`). +6. WER: `wer_plan.py` writes the segment list of every system; `wer_dec.py` decodes those segments; + `wer_report.py` prints `results/wer.md`. `mk_composite.py` builds the composite recordings (three TED + talks with 40 s of music, 30 s of noise and 40 s of vocal music inserted at pauses) and the plans with + argument `comp` give `results/wer_comp.md`. `wer_plan_forced.py` and `wer_report.py forced` give + `results/wer_forced.md`, the cost of a forced cut. + `wer_dec.py` needs a throwaway `parakeet-cli` patched to decode a given list of segments (it reads them + from the file named by the environment variable `PK_SEGMENTS` and writes the words to `PK_SEGOUT`). That + patch is not part of the repository, so this step cannot be repeated from the repository alone. +7. `verify_gate.py` checks the C++ run gate against the Python gate (`vr.gate_frames`) on the stored + probabilities. Build `gate_segtool.cpp` first; the command is in the docstring of the script. On the + 688 comparisons of the published run (all recordings, three detectors, two or three gates each) the regions + were identical, and the speech seconds the Redux head loses at 0.92 matched to the millisecond. + +## How the gate is defined here + +`vr.gate_frames(p, thr, med)` finds every run of consecutive frames with `p >= thr` in the detector's own +frames (80 ms or 32 ms), before any bridging, and keeps the run when the median of those frames is at least `med`. +This is the definition of `SegmenterOpts::run_gate` in `src/vad_segmenter.hpp`. + +## Results + +`results/frames.md` frame F1 of every system, tuned and untuned, with paired differences (sections A to F); +`options.md` option sweeps for Silero and the heads; `fusion_grid.md` the fusion rule grid; `extras.md` the +collar test and music by source; `seg.md` what the decoder receives (trim, cut policies); `wer.md`, +`wer_comp.md` and `wer_forced.md` word error rates; `tuned.json` the parameters chosen on the tuning split. diff --git a/scripts/vad_bench/real_recordings/dump_probs.py b/scripts/vad_bench/real_recordings/dump_probs.py new file mode 100644 index 0000000..1542b80 --- /dev/null +++ b/scripts/vad_bench/real_recordings/dump_probs.py @@ -0,0 +1,28 @@ +#!/usr/bin/env python3 +"""Run parakeet-cli vad --probabilities (clean master build) for every recording and detector; store probs (.npy) and the CLI's default speech regions (.json).""" +import glob, json, os, subprocess, sys +import numpy as np +from concurrent.futures import ThreadPoolExecutor +ROOT=os.environ.get("VAD_REAL_ROOT") or os.getcwd() +CLI_CLEAN=os.environ.get("PARAKEET_CLI", f"{ROOT}/parakeet-cli-clean"); MODELS=os.environ.get("VAD_REAL_MODELS", f"{ROOT}/models") +CLI=CLI_CLEAN; MOD={"silero":"silero-vad-f16","ultra":"ultra-vad-q8_0","redux":"redux-vad"} +os.makedirs(f"{ROOT}/probs",exist_ok=True) +jobs=[] +for pat in ("data/vox/*.json","data/ami/*.json","data/ava/*.json","data/musan/*.json","data/esc/*.json"): + for j in sorted(glob.glob(f"{ROOT}/{pat}")): + w=j[:-5]+".wav" + if not os.path.exists(w): continue + dom=pat.split("/")[1]; rid=os.path.basename(w)[:-4] + if dom=="musan": dom=rid.split("_")[0]; rid=rid + if dom in("vox","ami"): rid=f"{dom}_{rid}" + elif dom=="ava": rid=f"ava_{rid}" + else: rid=f"{dom}_{rid}" + for k in MOD: jobs.append((w,rid,k)) +def run(a): + w,rid,k=a; out=f"{ROOT}/probs/{rid}.{k}.npy" + if os.path.exists(out): return + r=subprocess.run([CLI,"vad","--model",f"{MODELS}/{MOD[k]}.gguf","--input",w,"--probabilities","--threads","2"],capture_output=True,text=True) + j=json.loads(r.stdout) + np.save(out,np.array(j["probabilities"],np.float32)); json.dump({"segments":j["segments"],"duration":j["duration"],"frame_sec":j["frame_sec"]},open(f"{ROOT}/probs/{rid}.{k}.json","w")) +with ThreadPoolExecutor(8) as ex: list(ex.map(run,jobs)) +print("done",len(jobs)) diff --git a/scripts/vad_bench/real_recordings/eval_frames.py b/scripts/vad_bench/real_recordings/eval_frames.py new file mode 100644 index 0000000..6885e8d --- /dev/null +++ b/scripts/vad_bench/real_recordings/eval_frames.py @@ -0,0 +1,53 @@ +#!/usr/bin/env python3 +"""Compute per-recording confusion counts (tp,fp,fn,tn on 10 ms cells) + number of speech regions for every system spec. -> results/counts.pkl""" +import os, sys, pickle, itertools, time +import numpy as np +from multiprocessing import Pool +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import vr, systems +THR = [0.02, 0.05, 0.075] + [round(x, 2) for x in np.arange(0.1, 0.951, 0.05)] +THR9 = [round(x, 1) for x in np.arange(0.1, 0.91, 0.1)] +HEADS = ("ultra", "redux") +def specs(): + S = [] + S += [("ref",)] + S += [("sil", t) for t in THR] + S += [("nat", "silero", t, ms, mp, pad) for t in (0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7) for ms in (0.1, 0.25, 0.5) for mp in (0.1, 0.2, 0.5) for pad in (0.0, 0.03, 0.1, 0.2)] + for h in HEADS: + S += [("head", h, t) for t in THR] + S += [("nat", h, 0.5, 0.1, 0.2, 0.0)] + S += [("or", h, a, b) for a in THR9 for b in THR9] + S += [("and", h, a, b) for a in THR9 for b in THR9] + S += [("mean", h, w, t) for w in (0.25, 0.5, 0.75) for t in THR] + S += [("two", h, ts, th, pre, po, Y) for ts in (0.1, 0.2, 0.3, 0.5, 0.7) for th in (0.3, 0.5, 0.7, 0.9) for pre in (0, 8, 16, 24, 32) for po in (0, 8, 16, 32) for Y in (0, 30, 60)] + S += [("gate", h, m) for m in (0.5, 0.6, 0.7, 0.8, 0.85, 0.9, 0.92, 0.94, 0.96, 0.98, 0.99)] + S += [("org", h, m) for m in (0.8, 0.92, 0.96)] + S += [("twog", h, m, pre, po, Y) for m in (0.92,) for pre, po, Y in ((16, 16, 30), (32, 16, 60), (16, 16, 0), (0, 0, 30), (8, 8, 30))] + return S +PPG = [(ms, mp, pad) for ms in (0.1, 0.25, 0.5) for mp in (0.1, 0.2, 0.5) for pad in (0.0, 0.03, 0.1, 0.2)] +def specs_pp(): + inner = [("sil", t) for t in (0.05, 0.1, 0.2, 0.3, 0.4, 0.5, 0.6, 0.7)] + for h in HEADS: + inner += [("head", h, t) for t in (0.3, 0.4, 0.5, 0.6, 0.7, 0.8, 0.9)] + inner += [("or", h, a, b) for a in (0.1, 0.3, 0.5) for b in (0.3, 0.5, 0.7, 0.9)] + inner += [("mean", h, w, t) for w in (0.25, 0.5, 0.75) for t in (0.2, 0.3, 0.4, 0.5)] + inner += [("two", h, ts, th, pre, po, Y) for ts in (0.1, 0.3, 0.5) for th in (0.3, 0.5, 0.9) for pre in (0, 16, 32) for po in (0, 16, 32) for Y in (0, 30, 60)] + inner += [("gate", h, m) for m in (0.5, 0.8, 0.9, 0.92, 0.96)] + inner += [("org", h, m) for m in (0.8, 0.92)] + return [("pp", ms, mp, pad, i) for i in inner for (ms, mp, pad) in PPG] +def work(meta): + r = systems.Rec(meta); ref = meta["ref"]; out = [] + for sp in SPECS: + m = systems.mask(r, sp) + s, e = vr.runs(m) + out.append(np.concatenate([vr.counts(m, ref), [len(s)]])) + return meta["id"], np.array(out, np.int64) +SPECS = specs() + specs_pp() +if __name__ == "__main__": + D = vr.load_recordings(); print(len(D), "recordings", len(SPECS), "specs", flush=True) + t = time.time() + with Pool(10) as p: res = dict(p.map(work, D, chunksize=1)) + os.makedirs(f"{vr.ROOT}/results", exist_ok=True) + meta = [dict(id=m["id"], domain=m["domain"], split=m["split"], dur=m["dur"], nonspeech=m["nonspeech"]) for m in D] + pickle.dump(dict(specs=SPECS, meta=meta, counts=np.stack([res[m["id"]] for m in D])), open(f"{vr.ROOT}/results/counts.pkl", "wb")) + print("done", time.time() - t) diff --git a/scripts/vad_bench/real_recordings/extras.py b/scripts/vad_bench/real_recordings/extras.py new file mode 100644 index 0000000..fa6f596 --- /dev/null +++ b/scripts/vad_bench/real_recordings/extras.py @@ -0,0 +1,75 @@ +#!/usr/bin/env python3 +"""Extras: (1) collar sensitivity of the frame metrics, (2) music false alarms by MUSAN source, (3) share of errors near reference boundaries.""" +import sys, os, json, glob, numpy as np +from multiprocessing import Pool +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import rl, vr, systems +from rl import * +HEADS = ("ultra", "redux") +def ppfam(kind, h=None): return [s for s in SP if s[0] == "pp" and s[4][0] == kind and (h is None or s[4][1] == h)] +TP = {"sil": tune(ppfam("sil"))} +for h in HEADS: + for k in ("head", "two", "gate", "org"): TP[(k, h)] = tune(ppfam(k, h)) +SPECS = [("nat", "silero", 0.5, 0.25, 0.1, 0.03), ("head", "ultra", 0.5), ("head", "redux", 0.5)] +for h in HEADS: SPECS += [("two", h, 0.5, 0.5, 16, 16, 30), ("gate", h, 0.92)] +SPECS += [TP["sil"]] + [TP[(k, h)] for h in HEADS for k in ("head", "two", "gate", "org")] +COLL = (0.0, 0.1, 0.25) +D = {m["id"]: m for m in vr.load_recordings()} +def bnd_dist(ref): + """distance in cells to the nearest reference transition""" + n = len(ref); t = np.flatnonzero(np.diff(ref.astype(np.int8)) != 0) + 0.5 + if len(t) == 0: return np.full(n, 1e9) + x = np.arange(n)[:, None] if False else np.arange(n) + j = np.searchsorted(t, x); d = np.full(n, 1e9) + l = np.where(j > 0, np.abs(x - t[np.maximum(j - 1, 0)]), 1e9); r = np.where(j < len(t), np.abs(t[np.minimum(j, len(t) - 1)] - x), 1e9) + return np.minimum(l, r) +def work(m): + if m["nonspeech"]: return m["id"], None + r = systems.Rec(m); ref = m["ref"]; dist = bnd_dist(ref) * vr.GR; out = [] + for sp in SPECS: + mk = systems.mask(r, sp); row = [] + for c in COLL: + keep = dist >= c if c > 0 else np.ones(len(ref), bool) + row.append(vr.counts(mk[keep], ref[keep])) + # errors: fraction of fp / fn cells within 0.1 and 0.25 s of a boundary + fp = mk & ~ref; fn = ~mk & ref + row.append(np.array([fp.sum(), (fp & (dist < 0.1)).sum(), (fp & (dist < 0.25)).sum(), fn.sum(), (fn & (dist < 0.1)).sum(), (fn & (dist < 0.25)).sum()])) + out.append(row) + return m["id"], out +if __name__ == "__main__": + ids = [m for m in D.values()] + with Pool(6) as p: res = dict(p.map(work, ids, chunksize=1)) + lines = ["# Extras", "", "## 1. Collar sensitivity (cells within the collar of any reference speech/non-speech transition are ignored). Pooled F1, bootstrap CI over recordings (stratified by domain).", ""] + lines.append("untuned systems on all recordings, tuned systems on the held-out split\n") + lines.append("| system | split | F1 no collar | F1 collar 0.1 s | F1 collar 0.25 s | FP cells within 0.1 s of a boundary | FN cells within 0.1 s | FP within 0.25 s | FN within 0.25 s |\n|---|---|---|---|---|---|---|---|---|") + rng = np.random.default_rng(3) + for si, sp in enumerate(SPECS): + tuned = sp in [TP[k] for k in TP] + split = "held" if tuned else None + grp = [np.array([i for i, m in enumerate(M) if m["domain"] == d and not m["nonspeech"] and (split is None or m["split"] == split)]) for d in SPEECH_DOM] + cells = [] + for ci in range(3): + pt = sum(np.sum([res[M[i]["id"]][si][ci] for i in g], 0) for g in grp) + bs = [] + for _ in range(500): + tot = 0 + for g in grp: + b = rng.choice(g, size=len(g)); tot = tot + np.sum([res[M[i]["id"]][si][ci] for i in b], 0) + bs.append(vr.prf(tot)[2]) + l, h = np.percentile(bs, [2.5, 97.5]); cells.append(f"{100*vr.prf(pt)[2]:.1f} [{100*l:.1f}, {100*h:.1f}]") + e = sum(np.sum([res[M[i]["id"]][si][3] for i in g], 0) for g in grp) + lines.append(f"| {sp} | {split or 'all'} | " + " | ".join(cells) + f" | {100*e[1]/e[0]:.0f}% | {100*e[4]/e[3]:.0f}% | {100*e[2]/e[0]:.0f}% | {100*e[5]/e[3]:.0f}% |") + # music by source + src = {} + for j in glob.glob(f"{vr.ROOT}/data/musan/music_*.json"): + src[os.path.basename(j)[:-5]] = json.load(open(j))["source"] + lines += ["", "## 2. Music false alarms by MUSAN source (seconds called speech per hour of audio)", "", "| system | " + " | ".join(sorted(set(src.values()))) + " |", "|---|" + "---|" * len(set(src.values()))] + for sp in [("nat", "silero", 0.5, 0.25, 0.1, 0.03), ("head", "ultra", 0.5), ("head", "redux", 0.5), ("two", "ultra", 0.5, 0.5, 16, 16, 30), ("two", "redux", 0.5, 0.5, 16, 16, 30), ("gate", "ultra", 0.92), ("gate", "redux", 0.92), ("gate", "redux", 0.98)]: + row = [] + for sname in sorted(set(src.values())): + ii = [i for i, m in enumerate(M) if m["domain"] == "music" and src[m["id"].replace("music_", "", 1)] == sname] + if not ii: row.append("-"); continue + k = IDX[sp]; fp = C[ii, k, 0] + C[ii, k, 1]; n = C[ii, k, :4].sum(1) + row.append(f"{fp.sum()/n.sum()*3600:.0f} (n={len(ii)} files, {n.sum()*0.01/3600:.2f} h)") + lines.append(f"| {sp} | " + " | ".join(row) + " |") + open(f"{vr.ROOT}/results/extras.md", "w").write("\n".join(lines) + "\n"); print("\n".join(lines)) diff --git a/scripts/vad_bench/real_recordings/fetch_ava.py b/scripts/vad_bench/real_recordings/fetch_ava.py new file mode 100644 index 0000000..754d92b --- /dev/null +++ b/scripts/vad_bench/real_recordings/fetch_ava.py @@ -0,0 +1,17 @@ +#!/usr/bin/env python3 +"""AVA-Speech (human labels, film audio) from nccratliri/vad-human-ava-speech: download N clips by hash selection, convert to 16k mono int16.""" +import json, os, sys, hashlib, urllib.request, subprocess +from huggingface_hub import HfApi +outd=sys.argv[1]; N=int(sys.argv[2]) +os.environ.setdefault("HF_HOME",os.path.join(os.environ.get("VAD_REAL_ROOT") or os.getcwd(),"hfhome")) +api=HfApi(); files=[f for f in api.list_repo_files("nccratliri/vad-human-ava-speech",repo_type="dataset") if f.endswith(".wav")] +files.sort(key=lambda f: hashlib.md5(f.encode()).hexdigest()) +for f in files[:N]: + rid=os.path.basename(f)[:-4] + for ext in ("wav","json"): + u=f"https://huggingface.co/datasets/nccratliri/vad-human-ava-speech/resolve/main/{f[:-3]}{ext}" + tmp=f"{outd}/{rid}.raw.{ext}" + urllib.request.urlretrieve(u,tmp) + subprocess.run(["ffmpeg","-y","-loglevel","error","-i",f"{outd}/{rid}.raw.wav","-ac","1","-ar","16000","-c:a","pcm_s16le",f"{outd}/{rid}.wav"],check=True) + os.remove(f"{outd}/{rid}.raw.wav"); os.rename(f"{outd}/{rid}.raw.json",f"{outd}/{rid}.json") + print(rid,flush=True) diff --git a/scripts/vad_bench/real_recordings/fetch_diar.py b/scripts/vad_bench/real_recordings/fetch_diar.py new file mode 100644 index 0000000..5740aad --- /dev/null +++ b/scripts/vad_bench/real_recordings/fetch_diar.py @@ -0,0 +1,21 @@ +#!/usr/bin/env python3 +"""Stream diarizers-community voxconverse / ami(sdm), keep every k-th recording up to a duration cap. Writes wav (16k mono int16) + turns json.""" +import io, json, os, sys, re +import numpy as np, soundfile as sf, librosa +ROOT=os.environ.get("VAD_REAL_ROOT") or os.getcwd() +os.environ.setdefault("HF_HOME",f"{ROOT}/hfhome") +from datasets import Audio, load_dataset +name, cfg, splits, k, cap_h, outd = sys.argv[1], sys.argv[2], sys.argv[3].split(","), int(sys.argv[4]), float(sys.argv[5]), sys.argv[6] +tot=0.0 +for split in splits: + ds = load_dataset(name, None if cfg=="-" else cfg, split=split, streaming=True).cast_column("audio", Audio(decode=False)) + for i, ex in enumerate(ds): + if i % k != (k//2): continue + if tot/3600 >= cap_h: break + y, sr = sf.read(io.BytesIO(ex["audio"]["bytes"]), dtype="float32") + if y.ndim > 1: y = y.mean(axis=1) + if sr != 16000: y = librosa.resample(y, orig_sr=sr, target_sr=16000) + rid=f"{split}_{i:04d}" + sf.write(f"{outd}/{rid}.wav", y, 16000, subtype="PCM_16") + json.dump({"id":rid,"split":split,"dur":len(y)/16000,"start":ex["timestamps_start"],"end":ex["timestamps_end"],"speakers":ex["speakers"]}, open(f"{outd}/{rid}.json","w")) + tot+=len(y)/16000; print(rid, round(len(y)/60,1), "min; total h", round(tot/3600,2), flush=True) diff --git a/scripts/vad_bench/real_recordings/fetch_nonspeech.py b/scripts/vad_bench/real_recordings/fetch_nonspeech.py new file mode 100644 index 0000000..cd2aa8e --- /dev/null +++ b/scripts/vad_bench/real_recordings/fetch_nonspeech.py @@ -0,0 +1,35 @@ +#!/usr/bin/env python3 +"""Non-speech audio: MUSAN music (every k-th file), MUSAN noise, ESC-50 (every 4th clip) -> 16k mono wav files (noise/esc concatenated into ~10 min files).""" +import io, json, os, sys +import numpy as np, soundfile as sf, librosa +ROOT=os.environ.get("VAD_REAL_ROOT") or os.getcwd() +os.environ.setdefault("HF_HOME",f"{ROOT}/hfhome") +from datasets import Audio, load_dataset +which=sys.argv[1]; outd=sys.argv[2] +def dec(ex): + y,sr=sf.read(io.BytesIO(ex["audio"]["bytes"]),dtype="float32") + if y.ndim>1: y=y.mean(axis=1) + if sr!=16000: y=librosa.resample(y,orig_sr=sr,target_sr=16000) + return y +if which=="music": + k=int(sys.argv[3]); cap=float(sys.argv[4]); tot=0; start=int(sys.argv[5]) if len(sys.argv)>5 else 0 + ds=load_dataset("corypaik/musan","music",split="train",streaming=True).cast_column("audio",Audio(decode=False)) + for i,ex in enumerate(ds): + if i=cap*3600: break + y=dec(ex); y=y[:int(600*16000)] # cap a track at 10 min + sf.write(f"{outd}/music_{i:04d}.wav",y,16000,subtype="PCM_16"); tot+=len(y)/16000 + json.dump({"src":ex["path"],"source":ex["source"]},open(f"{outd}/music_{i:04d}.json","w")); print("music",i,ex["source"],round(len(y)/60,1),flush=True) +else: + if which=="noise": ds=load_dataset("corypaik/musan","noise",split="train",streaming=True); k=1; cap=float(sys.argv[3]); tag="noise" + else: ds=load_dataset("ashraq/esc50",split="train",streaming=True); k=4; cap=float(sys.argv[3]); tag="esc" + ds=ds.cast_column("audio",Audio(decode=False)) + buf=[];n=0;tot=0;files=0;meta=[] + for i,ex in enumerate(ds): + if i%k!=0: continue + if tot>=cap*3600: break + y=dec(ex); buf.append(y); tot+=len(y)/16000; n+=len(y) + meta.append(ex.get("path") or ex.get("category")) + if n>=600*16000: + sf.write(f"{outd}/{tag}_{files:02d}.wav",np.concatenate(buf),16000,subtype="PCM_16"); json.dump(meta,open(f"{outd}/{tag}_{files:02d}.json","w")); files+=1;buf=[];n=0;meta=[];print(tag,files,round(tot/3600,2),flush=True) + if buf: sf.write(f"{outd}/{tag}_{files:02d}.wav",np.concatenate(buf),16000,subtype="PCM_16"); json.dump(meta,open(f"{outd}/{tag}_{files:02d}.json","w")) diff --git a/scripts/vad_bench/real_recordings/gate_segtool.cpp b/scripts/vad_bench/real_recordings/gate_segtool.cpp new file mode 100644 index 0000000..ee61486 --- /dev/null +++ b/scripts/vad_bench/real_recordings/gate_segtool.cpp @@ -0,0 +1,15 @@ +// Reads raw float32 probabilities and prints the regions of speech_regions or segment_by_vad, for verify_gate.py. +// usage: gate_segtool probs.f32 total_sec kind(head|silero) run_gate mode(speech|segments) +#include "vad_segmenter.hpp" +#include +#include +#include +#include +int main(int argc, char** argv) { + std::FILE* f = std::fopen(argv[1], "rb"); std::vector p; float v; while (std::fread(&v, 4, 1, f) == 1) p.push_back(v); std::fclose(f); + double total = std::atof(argv[2]); + pk::SegmenterOpts o = pk::default_segmenter_opts(!std::strcmp(argv[3], "silero") ? pk::VadKind::kSilero : pk::VadKind::kHead); + o.run_gate = (float)std::atof(argv[4]); + auto r = !std::strcmp(argv[5], "speech") ? pk::speech_regions(p, total, o) : pk::segment_by_vad(p, total, o); + for (auto& s : r) std::printf("%.6f %.6f\n", s.start, s.end); +} diff --git a/scripts/vad_bench/real_recordings/mk_composite.py b/scripts/vad_bench/real_recordings/mk_composite.py new file mode 100644 index 0000000..6a0a635 --- /dev/null +++ b/scripts/vad_bench/real_recordings/mk_composite.py @@ -0,0 +1,31 @@ +#!/usr/bin/env python3 +"""Composite long recordings: a real TED talk with 3 real non-speech clips (MUSAN music x2, noise x1; 40/30/40 s) inserted at Silero pauses near 25/50/75 percent. +The reference transcript is the talk's own transcript (inserted clips contain no scored speech), so every inserted word counts as an insertion.""" +import glob, json, os, numpy as np, soundfile as sf +import vr +ROOT = vr.ROOT +src = {os.path.basename(j)[:-5]: json.load(open(j))["source"] for j in glob.glob(f"{ROOT}/data/musan/music_*.json")} +pick = {"fma": sorted(k for k, v in src.items() if v == "fma")[0], "jamendo": sorted(k for k, v in src.items() if v == "jamendo")[1], "noise": "noise_01"} +print(pick) +def clip(name, sec, off=20): + y, sr = sf.read(f"{ROOT}/data/musan/{name}.wav", dtype="float32"); assert sr == 16000 + y = y[off * 16000: off * 16000 + sec * 16000]; return y +os.makedirs(f"{ROOT}/data/comp", exist_ok=True) +for t in ["GaryFlake-merged", "RobertGupta-merged", "EricMead_2009P_EricMead-merged"]: + y, _ = sf.read(f"{ROOT}/data/ted/{t}.wav", dtype="float32"); dur = len(y) / 16000 + P = np.load(f"{ROOT}/probs/ted_{t.replace('-merged','')}.silero.npy") + sil = P < 0.3; s, e = vr.runs(sil); mids = [(a + b) / 2 * 0.032 for a, b in zip(s, e) if (b - a) * 0.032 >= 0.4] + rms = np.sqrt((y ** 2).mean()) + ins = [] + for frac, key, sec in ((0.25, "fma", 40), (0.5, "noise", 30), (0.75, "jamendo", 40)): + c = clip(pick[key], sec); c = c * (0.7 * rms / (np.sqrt((c ** 2).mean()) + 1e-9)); c = np.clip(c, -1, 1) + fade = np.linspace(0, 1, 1600, dtype=np.float32); c[:1600] *= fade; c[-1600:] *= fade[::-1] + pos = min(mids, key=lambda m: abs(m - frac * dur)); ins.append((pos, c)) + out = []; last = 0.0 + for pos, c in sorted(ins, key=lambda x: x[0]): + out += [y[int(last * 16000):int(pos * 16000)], c]; last = pos + out.append(y[int(last * 16000):]); z = np.concatenate(out) + name = t.replace("-merged", "") + "+ins" + sf.write(f"{ROOT}/data/comp/{name}.wav", z, 16000, subtype="PCM_16") + open(f"{ROOT}/data/comp/{name}.txt", "w").write(open(f"{ROOT}/data/ted/{t}.txt").read()) + print(name, round(len(z) / 16000), "s", [round(p) for p, _ in ins]) diff --git a/scripts/vad_bench/real_recordings/ref_check.py b/scripts/vad_bench/real_recordings/ref_check.py new file mode 100644 index 0000000..b255aca --- /dev/null +++ b/scripts/vad_bench/real_recordings/ref_check.py @@ -0,0 +1,17 @@ +#!/usr/bin/env python3 +"""Reference sanity: frame energy (dB, 25 ms window / 10 ms hop) in cells that are (a) ref speech and detected by Silero .5, (b) ref speech missed by Silero, (c) ref non-speech and called speech by Redux head only, (d) ref non-speech and not detected. Per domain medians relative to the file's 10th percentile energy.""" +import numpy as np, soundfile as sf, vr, systems +D = vr.load_recordings(); rows = {} +for m in D: + if m["nonspeech"]: continue + y, _ = sf.read(m["wav"], dtype="float32"); n = len(m["ref"]); fr = np.lib.stride_tricks.sliding_window_view(np.pad(y, (0, 400)), 400)[::160][:n] + e = 10 * np.log10((fr ** 2).mean(1) + 1e-10); floor = np.percentile(e, 10); e = e - floor + r = systems.Rec(m); S = systems.mask(r, ("sil", 0.5)); H = systems.mask(r, ("head", "redux", 0.5)); ref = m["ref"] + for nm, sel in (("ref speech, Silero hit", ref & S), ("ref speech, Silero miss", ref & ~S), ("ref speech, Redux miss", ref & ~H), ("ref non-speech, Redux only", ~ref & H & ~S), ("ref non-speech, neither", ~ref & ~H & ~S)): + rows.setdefault((m["domain"], nm), []).append((sel.sum(), np.median(e[sel]) if sel.sum() else np.nan)) +print("| domain | cells | share of domain cells | median level above file noise floor (dB) |\n|---|---|---|---|") +for d in ("vox", "ami", "ava"): + tot = sum(m["ref"].size for m in D if m["domain"] == d) + for nm in ("ref speech, Silero hit", "ref speech, Silero miss", "ref speech, Redux miss", "ref non-speech, Redux only", "ref non-speech, neither"): + v = rows[(d, nm)]; n = sum(a for a, _ in v); med = np.nanmedian([b for a, b in v if a > 100]) + print(f"| {d} | {nm} | {100*n/tot:.1f}% | {med:.1f} |") diff --git a/scripts/vad_bench/real_recordings/report_frames.py b/scripts/vad_bench/real_recordings/report_frames.py new file mode 100644 index 0000000..487e372 --- /dev/null +++ b/scripts/vad_bench/real_recordings/report_frames.py @@ -0,0 +1,177 @@ +#!/usr/bin/env python3 +"""Frame-level tables (markdown) from results/counts.pkl: speech domains P/R/F1 with bootstrap CIs (recordings resampled within domain), tuned-on-tune/held-out, leave-one-domain-out, non-speech false alarms.""" +import sys, os, numpy as np, json +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import rl +from rl import * +HEADS = ("ultra", "redux") +out = [] +def P(*a): out.append(" ".join(str(x) for x in a)) +def groups(split=None, doms=SPEECH_DOM): return [sel((d,), split) for d in doms] +def cell(spec, g, which=2): + pt, bs = stat(spec, g, which); return fmt(pt, bs) +def rowtxt(name, spec, split=None): + cells = [cell(spec, [sel((d,), split)]) for d in SPEECH_DOM] + gall = groups(split) + pooled = cell(spec, gall, 2); pp = cell(spec, gall, 0); rr = cell(spec, gall, 1) + return f"| {name} | " + " | ".join(cells) + f" | {pooled} | {pp} | {rr} |" +HDR = "| system | vox F1 | ami F1 | ava F1 | pooled F1 | pooled P | pooled R |\n|---|---|---|---|---|---|---|" +# ---------------- tuning on the tune split (macro F1 over the three domains) +T = {} # thr/params only, default post +T["sil"] = tune([s for s in SP if s[0] == "sil"]) +T["silnat"] = tune([s for s in SP if s[0] == "nat" and s[1] == "silero"]) +for h in HEADS: + for k in ("head", "or", "and", "mean", "two"): + T[(k, h)] = tune([s for s in SP if s[0] == k and s[1] == h]) + T[("twog", h)] = tune([s for s in SP if s[0] == "twog" and s[1] == h]) + T[("gate", h)] = tune([s for s in SP if s[0] == "gate" and s[1] == h]) +def ppfam(kind, h=None): return [s for s in SP if s[0] == "pp" and s[4][0] == kind and (h is None or s[4][1] == h)] +TP = {} # params + post-processing (min speech, min pause, pad) tuned jointly +TP["sil"] = tune(ppfam("sil")) +for h in HEADS: + for k in ("head", "or", "mean", "two", "gate", "org"): + TP[(k, h)] = tune(ppfam(k, h)) +P("# Tuned parameters (chosen on the tune split only, objective = mean F1 of vox, ami, ava)") +for k, v in T.items(): P(f"- {k}: {v} (tune F1 {100*macro_f1(v,'tune'):.2f}, held F1 {100*macro_f1(v,'held'):.2f})") +P("\n# Tuned parameters incl. post-processing (ms, mp, pad), chosen on the tune split only") +for k, v in TP.items(): P(f"- {k}: {v} (tune F1 {100*macro_f1(v,'tune'):.2f}, held F1 {100*macro_f1(v,'held'):.2f})") +json.dump({str(k): list(map(lambda x: float(x) if isinstance(x, (float, np.floating)) else x, v)) for k, v in T.items()}, open(f"{rl.vr.ROOT}/results/tuned.json", "w")) +DEF = {"sil": ("sil", 0.5), "silnat": ("nat", "silero", 0.5, 0.25, 0.1, 0.03)} +FDEF = lambda h: ("two", h, 0.5, 0.5, 16, 16, 30) +def block(title, split, include_tuned): + P(f"\n## {title}\n"); P(HDR) + P(rowtxt("Silero, own defaults (thr .5, min speech .25, pause .1, pad .03)", DEF["silnat"], split)) + P(rowtxt("Silero thr .5, unified post", DEF["sil"], split)) + for h in HEADS: + P(rowtxt(f"{h} head thr .5 (own defaults)", ("head", h, 0.5), split)) + for h in HEADS: + P(rowtxt(f"OR(Silero .5, {h} .5)", ("or", h, 0.5, 0.5), split)) + P(rowtxt(f"AND(Silero .5, {h} .5)", ("and", h, 0.5, 0.5), split)) + P(rowtxt(f"mean(Silero, {h}) >= .5", ("mean", h, 0.5, 0.5), split)) + P(rowtxt(f"two-stage fusion {h}, defaults (ts .5, th .5, pre 160 ms, post 160 ms, fill 300 ms)", FDEF(h), split)) + for h in HEADS: + P(rowtxt(f"{h} head + median gate 0.92", ("gate", h, 0.92), split)) + P(rowtxt(f"two-stage fusion {h} with gated head (.92)", ("twog", h, 0.92, 16, 16, 30), split)) + P(rowtxt(f"OR(Silero .5, gated {h} .92)", ("org", h, 0.92), split)) + if include_tuned: + P(rowtxt(f"Silero thr tuned {T['sil'][1]}", T["sil"], split)) + P(rowtxt(f"Silero all options tuned {T['silnat'][2:]}", T["silnat"], split)) + for h in HEADS: + for k, nm in (("head", "thr"), ("or", "OR"), ("and", "AND"), ("mean", "mean"), ("two", "two-stage")): + P(rowtxt(f"{h} {nm} tuned {T[(k,h)][1:] if k!='head' else T[(k,h)][2]}", T[(k, h)], split)) + P(rowtxt(f"Silero thr + post tuned {TP['sil'][1:4]} {TP['sil'][4][1]}", TP["sil"], split)) + for h in HEADS: + for k, nm in (("head", "head thr"), ("or", "OR"), ("mean", "mean"), ("two", "two-stage fusion"), ("gate", "median gate"), ("org", "OR with gated head")): + P(rowtxt(f"{h} {nm}, params + post tuned {TP[(k,h)][1:4]} {TP[(k,h)][4][1:]}", TP[(k, h)], split)) +block("A. All recordings, untuned systems (no parameter was fitted to these; 9.9 h of speech-bearing audio)", None, False) +block("B. Held-out recordings only (vox test, ami test, ava hash split), untuned and tuned-on-tune systems", "held", True) +# ---------------- paired deltas +P("\n## C. Paired differences in F1 points (bootstrap over recordings, 95% CI), pooled and per domain\n") +P("| comparison | split | vox | ami | ava | pooled |\n|---|---|---|---|---|---|") +def drow(name, a, b, split): + cs = [] + for g in [[sel((d,), split)] for d in SPEECH_DOM] + [groups(split)]: + pt, bs = delta(a, b, g); cs.append(fmt(pt, bs, 100, 2)) + P(f"| {name} | {split or 'all'} | " + " | ".join(cs) + " |") +best_single_def = {} +for h in HEADS: + drow(f"fusion defaults {h} minus {h} head .5", FDEF(h), ("head", h, 0.5), None) + drow(f"fusion defaults {h} minus Silero own defaults", FDEF(h), DEF["silnat"], None) + drow(f"{h} head .5 minus Silero own defaults", ("head", h, 0.5), DEF["silnat"], None) + drow(f"gate .92 {h} minus {h} head .5", ("gate", h, 0.92), ("head", h, 0.5), None) +for h in HEADS: + drow(f"fusion tuned {h} (thr/params only, default post) minus {h} head tuned (thr only)", T[("two", h)], T[("head", h)], "held") + drow(f"fusion tuned {h} (thr/params only, default post) minus Silero thr tuned (default post)", T[("two", h)], T["sil"], "held") +P("\n### C2. Paired differences, everything tuned jointly with post-processing (held-out split)\n") +P("| comparison | split | vox | ami | ava | pooled |\n|---|---|---|---|---|---|") +for h in HEADS: + best = max([TP["sil"], TP[("head", h)]], key=lambda s: macro_f1(s, "tune")) + for k, nm in (("two", "fusion"), ("gate", "gated head"), ("org", "OR with gated head"), ("or", "OR"), ("mean", "mean")): + drow(f"{h} {nm} {TP[(k,h)][1:4]} minus best single detector tuned on tune ({'Silero' if best==TP['sil'] else h+' head'})", TP[(k, h)], best, "held") + drow(f"{h} head tuned minus Silero tuned", TP[("head", h)], TP["sil"], "held") + drow(f"{h} fusion minus Silero tuned", TP[("two", h)], TP["sil"], "held") +# ---------------- leave one domain out +P("\n## D. Leave-one-domain-out: parameters tuned on the other two domains (all their recordings), scored on the left-out domain (F1 and 95% CI)\n") +P("| system | vox (tuned on ami+ava) | ami (tuned on vox+ava) | ava (tuned on vox+ami) | mean |\n|---|---|---|---|---|") +def lodo_pick(cands, hold): + others = [d for d in SPEECH_DOM if d != hold] + def sc(s): + f = [] + for d in others: + ii = sel((d,)); c = C[ii, IDX[s], :4].sum(0); f.append(f1_of(c)[2]) + return np.mean(f) + return max(cands, key=sc) +fams = {"Silero thr": [s for s in SP if s[0] == "sil"], "Silero all options": ppfam("sil")} +for h in HEADS: + for k, nm in (("head", "head thr"), ("or", "OR"), ("mean", "mean"), ("two", "two-stage")): + fams[f"{h} {nm}"] = [s for s in SP if s[0] == k and s[1] == h] + fams[f"{h} {nm} + post"] = ppfam(k, h) + fams[f"{h} gate + post"] = ppfam("gate", h) +LODO = {} +for nm, cands in fams.items(): + cs = []; vals = []; picks = [] + for d in SPEECH_DOM: + b = lodo_pick(cands, d); picks.append(b); pt, bs = stat(b, [sel((d,))]); cs.append(fmt(pt, bs)); vals.append(pt) + LODO[nm] = picks + P(f"| {nm} | " + " | ".join(cs) + f" | {100*np.mean(vals):.1f} |") +P("\nPicks: " + "; ".join(f"{k}: {[tuple(x[1:]) if len(x)>2 else x for x in v]}" for k, v in LODO.items())) +def others_score(spec, hold): + f = [] + for d in SPEECH_DOM: + if d == hold: continue + ii = sel((d,)); c = C[ii, IDX[spec], :4].sum(0); f.append(f1_of(c)[2]) + return np.mean(f) +P("\nLODO paired deltas in F1 points (left-out domain; the best single detector is chosen by its score on the two tuning domains; CI over recordings of the left-out domain):\n") +P("| comparison | vox | ami | ava | mean of three |\n|---|---|---|---|---|") +for h in HEADS: + for fam in ("two-stage", "gate", "OR", "mean"): + key_a = f"{h} {fam} + post" if fam != "gate" else f"{h} gate + post" + if key_a not in LODO: continue + for withpost in (True,): + cs = []; vals = [] + for d_i, d in enumerate(SPEECH_DOM): + a = LODO[key_a][d_i] + singles = [LODO["Silero all options"][d_i], LODO[f"{h} head thr + post"][d_i]] + bsel = max(singles, key=lambda x: others_score(x, d)) + pt, bs = delta(a, bsel, [sel((d,))]); cs.append(fmt(pt, bs, 100, 2)); vals.append(pt * 100) + P(f"| {h} {fam} (+post) minus best single (+post) | " + " | ".join(cs) + f" | {np.mean(vals):.2f} |") + cs = []; vals = [] + for d_i, d in enumerate(SPEECH_DOM): + pt, bs = delta(LODO[f"{h} head thr + post"][d_i], LODO["Silero all options"][d_i], [sel((d,))]); cs.append(fmt(pt, bs, 100, 2)); vals.append(pt * 100) + P(f"| {h} head (+post) minus Silero (+post) | " + " | ".join(cs) + f" | {np.mean(vals):.2f} |") +# ---------------- gate sweep +P("\n## E. Median-probability gate sweep (head thr .5, keep a run if its median probability >= m; unified post), all recordings\n") +P("| head | m | vox F1 | ami F1 | ava F1 | pooled F1 | music FA s/h | noise FA s/h | esc FA s/h |\n|---|---|---|---|---|---|---|---|---|") +def fa_rate(spec, dom_list, g=None): + """speech seconds called per hour of audio, bootstrap over files (ratio of sums)""" + ii = sel(dom_list) if g is None else g; k = IDX[spec] + fp = (C[ii, k, 0] + C[ii, k, 1]); n = C[ii, k, :4].sum(1) # tp is 0 on non-speech recordings; fp = predicted cells + rng = np.random.default_rng(7); bi = rng.integers(0, len(ii), size=(2000, len(ii))) + est = fp.sum() * 0.01 / (n.sum() * 0.01 / 3600); bs = fp[bi].sum(1) / n[bi].sum(1) * 3600 + l, h = np.percentile(bs, [2.5, 97.5]); return f"{est:.0f} [{l:.0f}, {h:.0f}]" +for h in HEADS: + for m in (0.5, 0.6, 0.7, 0.8, 0.85, 0.9, 0.92, 0.94, 0.96, 0.98, 0.99): + sp = ("gate", h, m); g = groups(None) + P(f"| {h} | {m} | " + " | ".join(cell(sp, [sel((d,))]) for d in SPEECH_DOM) + f" | {cell(sp, g)} | " + " | ".join(fa_rate(sp, (d,)) for d in NS_DOM) + " |") +# ---------------- non-speech FA +P("\n## F. Non-speech audio: seconds called speech per hour of audio and speech regions per hour (95% CI over files; domains: music 16 files, noise 6, esc 5)\n") +P("| system | music s/h | noise s/h | esc s/h | pooled s/h | pooled regions/h |\n|---|---|---|---|---|---|") +def regions_rate(spec, doms): + ii = sel(doms); k = IDX[spec]; nr = C[ii, k, 4]; n = C[ii, k, :4].sum(1) + rng = np.random.default_rng(8); bi = rng.integers(0, len(ii), size=(2000, len(ii))) + est = nr.sum() / (n.sum() * 0.01 / 3600); bs = nr[bi].sum(1) / (n[bi].sum(1) * 0.01 / 3600) + l, h = np.percentile(bs, [2.5, 97.5]); return f"{est:.0f} [{l:.0f}, {h:.0f}]" +NSR = [("Silero own defaults", DEF["silnat"]), ("Silero thr .5 unified", DEF["sil"]), (f"Silero tuned thr {T['sil'][1]}", T["sil"]), (f"Silero all options tuned", T["silnat"])] +for h in HEADS: + NSR += [(f"{h} head .5", ("head", h, 0.5)), (f"{h} head tuned {T[('head',h)][2]}", T[("head", h)]), (f"OR(S .5,{h} .5)", ("or", h, 0.5, 0.5)), (f"AND(S .5,{h} .5)", ("and", h, 0.5, 0.5)), + (f"fusion defaults {h}", FDEF(h)), (f"fusion tuned {h} {T[('two',h)][2:]}", T[("two", h)]), (f"{h} gate .92", ("gate", h, 0.92)), (f"fusion with gated head {h}", ("twog", h, 0.92, 16, 16, 30)), + (f"OR(S .5, gated {h})", ("org", h, 0.92))] +NSR += [("Silero thr+post tuned " + str(TP['sil'][1:4]) + " thr " + str(TP['sil'][4][1]), TP["sil"])] +for h in HEADS: + for k, nm in (("head", "head thr"), ("two", "fusion"), ("gate", "median gate"), ("org", "OR with gated head")): + NSR.append((f"{h} {nm} thr/params+post tuned {TP[(k,h)][1:4]}", TP[(k, h)])) +for nm, sp in NSR: + P(f"| {nm} | " + " | ".join(fa_rate(sp, (d,)) for d in NS_DOM) + f" | {fa_rate(sp, NS_DOM)} | {regions_rate(sp, NS_DOM)} |") +# music by source +open(f"{rl.vr.ROOT}/results/frames.md", "w").write("\n".join(out) + "\n") +print("\n".join(out)) diff --git a/scripts/vad_bench/real_recordings/report_fusion_grid.py b/scripts/vad_bench/real_recordings/report_fusion_grid.py new file mode 100644 index 0000000..dd1c091 --- /dev/null +++ b/scripts/vad_bench/real_recordings/report_fusion_grid.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Fusion extension sizes: pooled F1 (all recordings, ts .5 th .5, default post) and non-speech FA, as a function of pre/post/fill. Shows which part of the fusion carries the gain.""" +import sys, os, numpy as np +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import rl +from rl import * +out = ["# Fusion rule, ts = th = .5, default post, all recordings: delta F1 (pp) versus Silero .5 with the same post, paired bootstrap over recordings; FA = seconds called speech per hour on music/noise/esc pooled", "", + "| head | pre cells (10 ms) | post cells | fill cells | pooled F1 | dF1 vs Silero .5 [95% CI] | non-speech FA s/h |", "|---|---|---|---|---|---|---|"] +g = [sel((d,)) for d in SPEECH_DOM] +def fa(s): + ii = sel(NS_DOM); k = IDX[s]; return (C[ii, k, 1].sum()) / (C[ii, k, :4].sum()) * 3600 +for h in ("ultra", "redux"): + for pre, po, Y in ((0, 0, 0), (16, 0, 0), (32, 0, 0), (0, 16, 0), (0, 32, 0), (0, 0, 30), (0, 0, 60), (16, 16, 30), (32, 16, 30), (32, 32, 60), (16, 16, 60)): + s = ("two", h, 0.5, 0.5, pre, po, Y); pt, bs = stat(s, g); d, db = delta(s, ("sil", 0.5), g) + out.append(f"| {h} | {pre} | {po} | {Y} | {fmt(pt, bs)} | {fmt(d, db, 100, 2)} | {fa(s):.0f} |") +open(f"{rl.vr.ROOT}/results/fusion_grid.md", "w").write("\n".join(out) + "\n"); print("\n".join(out)) diff --git a/scripts/vad_bench/real_recordings/report_options.py b/scripts/vad_bench/real_recordings/report_options.py new file mode 100644 index 0000000..bc92387 --- /dev/null +++ b/scripts/vad_bench/real_recordings/report_options.py @@ -0,0 +1,29 @@ +#!/usr/bin/env python3 +"""One-at-a-time option sweeps (no selection involved): Silero native options and head thresholds. All recordings. F1 per domain + pooled with CIs, FA on non-speech.""" +import sys, os, numpy as np +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import rl +from rl import * +out = ["# Option sweeps, all recordings, one option changed from the Silero defaults (thr .5, min speech .25 s, min pause/silence .1 s, pad .03 s)", "", + "| option | value | vox F1 | ami F1 | ava F1 | pooled F1 | pooled P | pooled R | non-speech FA s/h (music / noise / esc) |", "|---|---|---|---|---|---|---|---|---|"] +def fa(spec, d): + ii = sel((d,)); k = IDX[spec]; fp = C[ii, k, 1]; n = C[ii, k, :4].sum(1); return fp.sum() / n.sum() * 3600 +def row(opt, val, spec): + cells = [fmt(*stat(spec, [sel((d,))])) for d in SPEECH_DOM]; g = [sel((d,)) for d in SPEECH_DOM] + out.append(f"| {opt} | {val} | " + " | ".join(cells) + f" | {fmt(*stat(spec, g))} | {fmt(*stat(spec, g, 0))} | {fmt(*stat(spec, g, 1))} | " + " / ".join(f"{fa(spec, d):.0f}" for d in NS_DOM) + " |") +D0 = (0.5, 0.25, 0.1, 0.03) +sp = lambda t=D0[0], ms=D0[1], mp=D0[2], pad=D0[3]: ("nat", "silero", t, ms, mp, pad) +row("(defaults)", "-", sp()) +for t in (0.1, 0.2, 0.3, 0.4, 0.6, 0.7): row("threshold", t, sp(t=t)) +for ms in (0.1, 0.5): row("min speech s", ms, sp(ms=ms)) +for mp in (0.2, 0.5): row("min silence s (pause)", mp, sp(mp=mp)) +for pad in (0.0, 0.1, 0.2): row("speech pad s", pad, sp(pad=pad)) +out += ["", "## Silero threshold with the unified post (ms .1, pause .2, pad 0), thresholds below .1", "", "| thr | vox F1 | ami F1 | ava F1 | pooled F1 | nonspeech FA s/h |", "|---|---|---|---|---|---|"] +for t in (0.02, 0.05, 0.075, 0.1, 0.2): + s = ("sil", t); out.append(f"| {t} | " + " | ".join(fmt(*stat(s, [sel((d,))])) for d in SPEECH_DOM) + f" | {fmt(*stat(s, [sel((d,)) for d in SPEECH_DOM]))} | " + " / ".join(f"{fa(s, d):.0f}" for d in NS_DOM) + " |") +out += ["", "## Head thresholds (unified post: bridge .1, min speech .1, pause .2, pad 0), all recordings", "", "| head | thr | vox F1 | ami F1 | ava F1 | pooled F1 | pooled P | pooled R | non-speech FA s/h (music / noise / esc) |", "|---|---|---|---|---|---|---|---|---|"] +for h in ("ultra", "redux"): + for t in (0.3, 0.5, 0.7, 0.9, 0.95): + s = ("head", h, float(t)); g = [sel((d,)) for d in SPEECH_DOM] + out.append(f"| {h} | {t} | " + " | ".join(fmt(*stat(s, [sel((d,))])) for d in SPEECH_DOM) + f" | {fmt(*stat(s, g))} | {fmt(*stat(s, g, 0))} | {fmt(*stat(s, g, 1))} | " + " / ".join(f"{fa(s, d):.0f}" for d in NS_DOM) + " |") +open(f"{rl.vr.ROOT}/results/options.md", "w").write("\n".join(out) + "\n"); print("\n".join(out)) diff --git a/scripts/vad_bench/real_recordings/results/extras.md b/scripts/vad_bench/real_recordings/results/extras.md new file mode 100644 index 0000000..d1ef414 --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/extras.md @@ -0,0 +1,37 @@ +# Extras + +## 1. Collar sensitivity (cells within the collar of any reference speech/non-speech transition are ignored). Pooled F1, bootstrap CI over recordings (stratified by domain). + +untuned systems on all recordings, tuned systems on the held-out split + +| system | split | F1 no collar | F1 collar 0.1 s | F1 collar 0.25 s | FP cells within 0.1 s of a boundary | FN cells within 0.1 s | FP within 0.25 s | FN within 0.25 s | +|---|---|---|---|---|---|---|---|---| +| ('nat', 'silero', 0.5, 0.25, 0.1, 0.03) | all | 90.6 [88.3, 92.1] | 91.3 [89.0, 92.9] | 91.7 [89.4, 93.5] | 40% | 6% | 62% | 12% | +| ('head', 'ultra', 0.5) | all | 90.8 [89.2, 92.0] | 91.5 [90.3, 92.8] | 91.9 [90.5, 93.1] | 14% | 7% | 24% | 12% | +| ('head', 'redux', 0.5) | all | 92.8 [91.6, 93.7] | 93.5 [92.3, 94.3] | 93.9 [92.8, 94.7] | 18% | 7% | 32% | 11% | +| ('two', 'ultra', 0.5, 0.5, 16, 16, 30) | all | 92.4 [90.5, 93.8] | 93.1 [91.3, 94.5] | 93.6 [91.9, 95.1] | 36% | 7% | 59% | 12% | +| ('gate', 'ultra', 0.92) | all | 90.2 [88.6, 91.9] | 91.0 [89.3, 92.5] | 91.5 [89.7, 93.1] | 31% | 6% | 52% | 12% | +| ('two', 'redux', 0.5, 0.5, 16, 16, 30) | all | 92.9 [91.2, 94.2] | 93.7 [92.0, 95.1] | 94.2 [92.5, 95.5] | 37% | 7% | 62% | 12% | +| ('gate', 'redux', 0.92) | all | 92.5 [90.8, 93.7] | 93.2 [92.0, 94.3] | 93.8 [92.4, 94.8] | 40% | 6% | 67% | 12% | +| ('pp', 0.25, 0.5, 0.1, ('sil', 0.1)) | held | 93.4 [91.2, 94.8] | 94.2 [92.3, 95.6] | 95.1 [93.2, 96.4] | 23% | 8% | 48% | 15% | +| ('pp', 0.25, 0.5, 0.2, ('head', 'ultra', 0.8)) | held | 91.5 [89.4, 93.1] | 92.4 [90.4, 94.0] | 93.2 [91.2, 94.7] | 20% | 5% | 41% | 11% | +| ('pp', 0.5, 0.5, 0.03, ('two', 'ultra', 0.1, 0.3, 32, 0, 0)) | held | 93.5 [91.5, 95.1] | 94.3 [92.3, 95.8] | 95.2 [93.4, 96.6] | 24% | 8% | 48% | 15% | +| ('pp', 0.25, 0.5, 0.1, ('gate', 'ultra', 0.8)) | held | 91.8 [89.8, 93.4] | 92.6 [90.4, 94.2] | 93.4 [91.3, 94.9] | 20% | 6% | 39% | 12% | +| ('pp', 0.1, 0.5, 0.1, ('org', 'ultra', 0.92)) | held | 93.2 [91.0, 94.9] | 94.1 [91.7, 95.6] | 94.9 [92.9, 96.5] | 27% | 6% | 53% | 13% | +| ('pp', 0.5, 0.5, 0.1, ('head', 'redux', 0.6)) | held | 93.3 [91.4, 94.6] | 94.2 [92.5, 95.3] | 95.0 [93.4, 96.2] | 23% | 7% | 45% | 15% | +| ('pp', 0.5, 0.5, 0.03, ('two', 'redux', 0.1, 0.5, 32, 16, 30)) | held | 93.8 [92.0, 95.4] | 94.7 [92.7, 96.0] | 95.6 [93.9, 96.9] | 24% | 9% | 49% | 17% | +| ('pp', 0.25, 0.5, 0.03, ('gate', 'redux', 0.8)) | held | 93.9 [92.3, 95.1] | 94.8 [93.0, 95.8] | 95.5 [93.9, 96.6] | 27% | 8% | 52% | 15% | +| ('pp', 0.25, 0.5, 0.03, ('org', 'redux', 0.8)) | held | 94.2 [92.3, 95.5] | 95.1 [93.7, 96.2] | 95.8 [94.2, 96.8] | 26% | 9% | 50% | 15% | + +## 2. Music false alarms by MUSAN source (seconds called speech per hour of audio) + +| system | fma | fma-western-art | hd-classical | jamendo | rfm | +|---|---|---|---|---|---| +| ('nat', 'silero', 0.5, 0.25, 0.1, 0.03) | 0 (n=3 files, 0.11 h) | 0 (n=3 files, 0.16 h) | 0 (n=1 files, 0.05 h) | 294 (n=6 files, 0.42 h) | 5 (n=3 files, 0.18 h) | +| ('head', 'ultra', 0.5) | 2655 (n=3 files, 0.11 h) | 2222 (n=3 files, 0.16 h) | 907 (n=1 files, 0.05 h) | 3066 (n=6 files, 0.42 h) | 3457 (n=3 files, 0.18 h) | +| ('head', 'redux', 0.5) | 2257 (n=3 files, 0.11 h) | 1627 (n=3 files, 0.16 h) | 115 (n=1 files, 0.05 h) | 2214 (n=6 files, 0.42 h) | 2860 (n=3 files, 0.18 h) | +| ('two', 'ultra', 0.5, 0.5, 16, 16, 30) | 2 (n=3 files, 0.11 h) | 0 (n=3 files, 0.16 h) | 0 (n=1 files, 0.05 h) | 355 (n=6 files, 0.42 h) | 13 (n=3 files, 0.18 h) | +| ('two', 'redux', 0.5, 0.5, 16, 16, 30) | 0 (n=3 files, 0.11 h) | 0 (n=3 files, 0.16 h) | 0 (n=1 files, 0.05 h) | 355 (n=6 files, 0.42 h) | 10 (n=3 files, 0.18 h) | +| ('gate', 'ultra', 0.92) | 1294 (n=3 files, 0.11 h) | 1052 (n=3 files, 0.16 h) | 46 (n=1 files, 0.05 h) | 1605 (n=6 files, 0.42 h) | 2864 (n=3 files, 0.18 h) | +| ('gate', 'redux', 0.92) | 840 (n=3 files, 0.11 h) | 215 (n=3 files, 0.16 h) | 0 (n=1 files, 0.05 h) | 709 (n=6 files, 0.42 h) | 940 (n=3 files, 0.18 h) | +| ('gate', 'redux', 0.98) | 8 (n=3 files, 0.11 h) | 0 (n=3 files, 0.16 h) | 0 (n=1 files, 0.05 h) | 163 (n=6 files, 0.42 h) | 165 (n=3 files, 0.18 h) | diff --git a/scripts/vad_bench/real_recordings/results/frames.md b/scripts/vad_bench/real_recordings/results/frames.md new file mode 100644 index 0000000..e2015ff --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/frames.md @@ -0,0 +1,244 @@ +# Tuned parameters (chosen on the tune split only, objective = mean F1 of vox, ami, ava) +- sil: ('sil', 0.05) (tune F1 92.20, held F1 91.67) +- silnat: ('nat', 'silero', 0.1, 0.25, 0.5, 0.1) (tune F1 93.30, held F1 92.91) +- ('head', 'ultra'): ('head', 'ultra', np.float64(0.35)) (tune F1 89.54, held F1 89.01) +- ('or', 'ultra'): ('or', 'ultra', np.float64(0.1), np.float64(0.9)) (tune F1 92.12, held F1 91.33) +- ('and', 'ultra'): ('and', 'ultra', np.float64(0.1), np.float64(0.1)) (tune F1 91.78, held F1 90.76) +- ('mean', 'ultra'): ('mean', 'ultra', 0.75, np.float64(0.25)) (tune F1 92.12, held F1 90.82) +- ('two', 'ultra'): ('two', 'ultra', 0.1, 0.3, 32, 8, 30) (tune F1 92.73, held F1 91.92) +- ('twog', 'ultra'): ('twog', 'ultra', 0.92, 32, 16, 60) (tune F1 91.28, held F1 89.94) +- ('gate', 'ultra'): ('gate', 'ultra', 0.85) (tune F1 90.37, held F1 88.58) +- ('head', 'redux'): ('head', 'redux', np.float64(0.6)) (tune F1 91.69, held F1 90.97) +- ('or', 'redux'): ('or', 'redux', np.float64(0.1), np.float64(0.8)) (tune F1 92.68, held F1 91.90) +- ('and', 'redux'): ('and', 'redux', np.float64(0.1), np.float64(0.1)) (tune F1 91.91, held F1 91.00) +- ('mean', 'redux'): ('mean', 'redux', 0.75, np.float64(0.2)) (tune F1 92.83, held F1 91.94) +- ('two', 'redux'): ('two', 'redux', 0.1, 0.3, 32, 8, 30) (tune F1 92.96, held F1 92.28) +- ('twog', 'redux'): ('twog', 'redux', 0.92, 32, 16, 60) (tune F1 91.60, held F1 90.88) +- ('gate', 'redux'): ('gate', 'redux', 0.8) (tune F1 92.41, held F1 91.81) + +# Tuned parameters incl. post-processing (ms, mp, pad), chosen on the tune split only +- sil: ('pp', 0.25, 0.5, 0.1, ('sil', 0.1)) (tune F1 93.32, held F1 92.91) +- ('head', 'ultra'): ('pp', 0.25, 0.5, 0.2, ('head', 'ultra', 0.8)) (tune F1 91.39, held F1 90.85) +- ('or', 'ultra'): ('pp', 0.5, 0.5, 0.1, ('or', 'ultra', 0.1, 0.9)) (tune F1 93.52, held F1 92.94) +- ('mean', 'ultra'): ('pp', 0.1, 0.5, 0.2, ('mean', 'ultra', 0.5, 0.5)) (tune F1 93.48, held F1 92.94) +- ('two', 'ultra'): ('pp', 0.5, 0.5, 0.03, ('two', 'ultra', 0.1, 0.3, 32, 0, 0)) (tune F1 93.58, held F1 92.98) +- ('gate', 'ultra'): ('pp', 0.25, 0.5, 0.1, ('gate', 'ultra', 0.8)) (tune F1 91.71, held F1 91.15) +- ('org', 'ultra'): ('pp', 0.1, 0.5, 0.1, ('org', 'ultra', 0.92)) (tune F1 93.29, held F1 92.59) +- ('head', 'redux'): ('pp', 0.5, 0.5, 0.1, ('head', 'redux', 0.6)) (tune F1 93.29, held F1 92.93) +- ('or', 'redux'): ('pp', 0.25, 0.5, 0.1, ('or', 'redux', 0.1, 0.9)) (tune F1 93.90, held F1 93.52) +- ('mean', 'redux'): ('pp', 0.5, 0.5, 0.1, ('mean', 'redux', 0.75, 0.2)) (tune F1 94.04, held F1 93.61) +- ('two', 'redux'): ('pp', 0.5, 0.5, 0.03, ('two', 'redux', 0.1, 0.5, 32, 16, 30)) (tune F1 93.75, held F1 93.41) +- ('gate', 'redux'): ('pp', 0.25, 0.5, 0.03, ('gate', 'redux', 0.8)) (tune F1 93.58, held F1 93.47) +- ('org', 'redux'): ('pp', 0.25, 0.5, 0.03, ('org', 'redux', 0.8)) (tune F1 93.93, held F1 93.83) + +## A. All recordings, untuned systems (no parameter was fitted to these; 9.9 h of speech-bearing audio) + +| system | vox F1 | ami F1 | ava F1 | pooled F1 | pooled P | pooled R | +|---|---|---|---|---|---|---| +| Silero, own defaults (thr .5, min speech .25, pause .1, pad .03) | 96.4 [95.4, 97.3] | 86.7 [78.9, 91.3] | 84.7 [81.6, 87.3] | 90.6 [88.4, 92.3] | 97.8 [97.3, 98.2] | 84.4 [80.7, 87.4] | +| Silero thr .5, unified post | 96.4 [95.4, 97.3] | 86.5 [78.8, 91.1] | 84.2 [80.9, 86.8] | 90.4 [88.2, 92.2] | 97.9 [97.5, 98.4] | 83.9 [80.2, 87.0] | +| ultra head thr .5 (own defaults) | 96.4 [95.2, 97.5] | 89.7 [86.0, 92.0] | 82.7 [78.8, 85.2] | 90.8 [89.3, 92.0] | 91.3 [89.1, 92.8] | 90.4 [88.7, 92.0] | +| redux head thr .5 (own defaults) | 97.3 [96.5, 98.1] | 92.2 [89.2, 94.1] | 85.4 [82.7, 87.6] | 92.8 [91.6, 93.7] | 92.8 [91.3, 94.1] | 92.7 [91.4, 93.8] | +| OR(Silero .5, ultra .5) | 96.7 [95.6, 97.7] | 90.9 [87.0, 93.3] | 83.3 [79.4, 85.9] | 91.5 [90.0, 92.8] | 91.2 [89.0, 92.7] | 91.9 [90.1, 93.4] | +| AND(Silero .5, ultra .5) | 96.0 [94.9, 97.0] | 85.0 [77.2, 89.6] | 83.3 [80.1, 86.0] | 89.6 [87.4, 91.3] | 98.2 [97.7, 98.6] | 82.3 [78.8, 85.3] | +| mean(Silero, ultra) >= .5 | 97.0 [96.1, 97.8] | 88.9 [83.7, 92.1] | 86.7 [84.3, 88.7] | 91.9 [90.3, 93.2] | 97.2 [96.6, 97.7] | 87.2 [84.6, 89.4] | +| two-stage fusion ultra, defaults (ts .5, th .5, pre 160 ms, post 160 ms, fill 300 ms) | 97.4 [96.5, 98.2] | 89.2 [82.9, 93.0] | 87.2 [84.6, 89.3] | 92.4 [90.5, 93.8] | 96.9 [96.2, 97.4] | 88.2 [85.0, 90.9] | +| OR(Silero .5, redux .5) | 97.5 [96.6, 98.2] | 92.5 [89.4, 94.3] | 85.6 [82.9, 87.8] | 93.0 [91.7, 93.9] | 92.6 [91.1, 93.9] | 93.3 [91.9, 94.4] | +| AND(Silero .5, redux .5) | 96.2 [95.2, 97.2] | 86.2 [78.4, 90.8] | 83.8 [80.5, 86.4] | 90.1 [88.0, 91.9] | 98.2 [97.8, 98.6] | 83.3 [79.6, 86.3] | +| mean(Silero, redux) >= .5 | 97.2 [96.4, 97.9] | 90.5 [86.1, 93.1] | 86.9 [84.3, 89.0] | 92.6 [91.2, 93.8] | 97.3 [96.8, 97.8] | 88.3 [85.9, 90.3] | +| two-stage fusion redux, defaults (ts .5, th .5, pre 160 ms, post 160 ms, fill 300 ms) | 97.6 [96.8, 98.3] | 90.4 [84.4, 93.8] | 87.5 [84.9, 89.7] | 92.9 [91.1, 94.3] | 96.8 [96.2, 97.3] | 89.3 [86.2, 91.8] | +| ultra head + median gate 0.92 | 96.4 [95.2, 97.4] | 85.4 [79.6, 89.3] | 85.4 [83.2, 87.5] | 90.2 [88.4, 91.7] | 96.5 [95.8, 97.2] | 84.7 [81.8, 87.3] | +| two-stage fusion ultra with gated head (.92) | 97.4 [96.5, 98.2] | 88.7 [82.0, 92.7] | 87.1 [84.4, 89.3] | 92.2 [90.3, 93.7] | 97.2 [96.6, 97.7] | 87.7 [84.3, 90.5] | +| OR(Silero .5, gated ultra .92) | 97.3 [96.4, 98.1] | 89.9 [84.7, 93.1] | 88.0 [86.0, 89.8] | 92.7 [91.1, 94.0] | 96.1 [95.4, 96.8] | 89.5 [86.8, 91.7] | +| redux head + median gate 0.92 | 96.9 [95.8, 97.8] | 90.8 [86.6, 93.3] | 86.2 [83.2, 88.6] | 92.5 [91.0, 93.7] | 97.0 [96.5, 97.5] | 88.3 [85.9, 90.4] | +| two-stage fusion redux with gated head (.92) | 97.5 [96.7, 98.3] | 90.1 [83.8, 93.6] | 87.3 [84.5, 89.6] | 92.7 [90.9, 94.2] | 97.0 [96.5, 97.5] | 88.8 [85.6, 91.4] | +| OR(Silero .5, gated redux .92) | 97.6 [96.8, 98.3] | 91.9 [88.0, 94.2] | 88.4 [85.9, 90.5] | 93.6 [92.3, 94.7] | 96.6 [96.0, 97.2] | 90.8 [88.5, 92.7] | + +## B. Held-out recordings only (vox test, ami test, ava hash split), untuned and tuned-on-tune systems + +| system | vox F1 | ami F1 | ava F1 | pooled F1 | pooled P | pooled R | +|---|---|---|---|---|---|---| +| Silero, own defaults (thr .5, min speech .25, pause .1, pad .03) | 94.9 [93.0, 96.2] | 82.2 [60.4, 90.1] | 86.6 [84.8, 88.5] | 89.1 [85.7, 91.5] | 97.2 [96.3, 98.0] | 82.2 [76.7, 86.5] | +| Silero thr .5, unified post | 94.8 [92.9, 96.2] | 81.8 [59.9, 90.1] | 86.2 [84.3, 88.0] | 88.8 [85.4, 91.3] | 97.3 [96.4, 98.1] | 81.7 [76.1, 86.0] | +| ultra head thr .5 (own defaults) | 95.0 [92.5, 96.8] | 86.8 [76.8, 91.8] | 84.3 [82.3, 86.3] | 89.3 [86.9, 91.1] | 89.6 [86.6, 92.1] | 89.0 [86.1, 92.1] | +| redux head thr .5 (own defaults) | 96.4 [94.7, 97.8] | 89.8 [81.5, 92.8] | 86.6 [84.0, 89.1] | 91.4 [89.2, 93.0] | 91.3 [88.4, 93.5] | 91.5 [89.2, 93.5] | +| OR(Silero .5, ultra .5) | 95.3 [92.9, 97.0] | 88.0 [77.7, 92.6] | 85.2 [83.0, 87.2] | 90.0 [87.6, 91.8] | 89.5 [86.5, 92.0] | 90.5 [87.7, 93.3] | +| AND(Silero .5, ultra .5) | 94.4 [92.4, 96.0] | 80.3 [58.5, 89.2] | 85.1 [83.2, 87.0] | 87.9 [84.5, 90.5] | 97.6 [96.8, 98.4] | 80.0 [74.6, 84.4] | +| mean(Silero, ultra) >= .5 | 95.7 [94.4, 96.9] | 85.0 [70.5, 91.6] | 87.8 [85.9, 89.5] | 90.4 [88.0, 92.4] | 96.4 [95.3, 97.5] | 85.1 [81.1, 88.7] | +| two-stage fusion ultra, defaults (ts .5, th .5, pre 160 ms, post 160 ms, fill 300 ms) | 96.3 [94.5, 97.6] | 85.3 [67.5, 92.1] | 88.6 [86.8, 90.4] | 91.1 [88.1, 93.2] | 96.2 [95.0, 97.2] | 86.5 [81.6, 90.4] | +| OR(Silero .5, redux .5) | 96.5 [94.8, 97.8] | 90.0 [81.6, 93.1] | 86.9 [84.4, 89.5] | 91.6 [89.4, 93.2] | 91.0 [88.1, 93.3] | 92.2 [89.8, 94.2] | +| AND(Silero .5, redux .5) | 94.6 [92.7, 96.1] | 81.5 [59.6, 89.6] | 85.7 [84.0, 87.4] | 88.5 [85.1, 91.0] | 97.7 [96.9, 98.4] | 80.9 [75.4, 85.1] | +| mean(Silero, redux) >= .5 | 96.0 [94.8, 97.1] | 87.4 [75.5, 92.2] | 88.4 [86.5, 90.0] | 91.4 [89.1, 93.0] | 96.7 [95.7, 97.6] | 86.6 [82.8, 89.5] | +| two-stage fusion redux, defaults (ts .5, th .5, pre 160 ms, post 160 ms, fill 300 ms) | 96.6 [94.9, 97.8] | 86.9 [70.0, 92.6] | 89.1 [87.3, 90.9] | 91.7 [88.9, 93.7] | 96.1 [95.0, 97.1] | 87.7 [82.9, 91.4] | +| ultra head + median gate 0.92 | 95.2 [93.1, 97.2] | 79.7 [64.0, 91.3] | 86.0 [84.2, 87.6] | 88.4 [85.6, 91.1] | 96.0 [94.7, 97.0] | 82.0 [77.6, 86.8] | +| two-stage fusion ultra with gated head (.92) | 96.3 [94.5, 97.6] | 84.5 [65.3, 92.0] | 88.6 [86.9, 90.3] | 90.9 [87.8, 93.2] | 96.5 [95.5, 97.4] | 85.8 [80.7, 89.9] | +| OR(Silero .5, gated ultra .92) | 96.2 [94.7, 97.6] | 86.0 [70.6, 93.2] | 89.1 [87.8, 90.3] | 91.4 [88.8, 93.3] | 95.5 [94.1, 96.6] | 87.6 [83.4, 91.3] | +| redux head + median gate 0.92 | 95.7 [93.6, 97.5] | 88.2 [76.5, 92.6] | 88.0 [86.4, 89.5] | 91.3 [88.9, 93.1] | 96.6 [95.7, 97.4] | 86.5 [82.8, 89.5] | +| two-stage fusion redux with gated head (.92) | 96.6 [94.8, 97.8] | 86.5 [69.0, 92.3] | 89.0 [87.3, 90.8] | 91.6 [88.7, 93.6] | 96.5 [95.4, 97.3] | 87.2 [82.2, 90.9] | +| OR(Silero .5, gated redux .92) | 96.7 [95.4, 97.8] | 89.2 [77.7, 93.7] | 90.1 [88.4, 91.6] | 92.6 [90.5, 94.2] | 96.0 [94.9, 97.0] | 89.5 [85.8, 92.3] | +| Silero thr tuned 0.05 | 96.4 [95.0, 97.4] | 90.4 [81.2, 93.8] | 88.2 [85.6, 90.9] | 92.1 [90.0, 93.7] | 92.2 [90.1, 93.9] | 92.0 [88.7, 94.4] | +| Silero all options tuned (0.1, 0.25, 0.5, 0.1) | 97.3 [95.7, 98.4] | 91.5 [82.3, 94.8] | 90.0 [87.5, 92.5] | 93.4 [91.2, 94.9] | 93.3 [91.5, 94.8] | 93.4 [90.1, 95.8] | +| ultra thr tuned 0.35 | 95.2 [92.4, 97.0] | 88.1 [78.9, 91.9] | 83.8 [81.1, 86.2] | 89.4 [86.9, 91.2] | 87.6 [84.0, 90.5] | 91.3 [88.8, 93.9] | +| ultra OR tuned ('ultra', np.float64(0.1), np.float64(0.9)) | 96.1 [94.9, 97.2] | 89.4 [79.0, 93.2] | 88.5 [86.9, 90.2] | 91.9 [89.8, 93.3] | 93.2 [91.4, 94.7] | 90.6 [87.3, 93.2] | +| ultra AND tuned ('ultra', np.float64(0.1), np.float64(0.1)) | 96.0 [94.8, 97.1] | 88.0 [76.8, 92.1] | 88.2 [85.9, 90.4] | 91.4 [89.2, 93.0] | 94.7 [93.1, 96.1] | 88.4 [84.7, 91.3] | +| ultra mean tuned ('ultra', 0.75, np.float64(0.25)) | 96.0 [94.7, 97.2] | 87.8 [76.5, 92.4] | 88.7 [86.5, 90.6] | 91.5 [89.3, 93.2] | 95.0 [93.4, 96.3] | 88.3 [84.7, 91.3] | +| ultra two-stage tuned ('ultra', 0.1, 0.3, 32, 8, 30) | 96.6 [95.1, 97.9] | 90.0 [80.7, 93.1] | 89.2 [86.7, 91.6] | 92.4 [90.4, 94.0] | 93.0 [91.0, 94.6] | 91.9 [88.8, 94.5] | +| redux thr tuned 0.6 | 96.3 [94.6, 97.5] | 89.5 [81.1, 92.3] | 87.1 [85.0, 89.2] | 91.5 [89.5, 93.0] | 93.1 [90.8, 94.9] | 89.9 [87.3, 92.1] | +| redux OR tuned ('redux', np.float64(0.1), np.float64(0.8)) | 96.4 [95.2, 97.4] | 90.3 [81.7, 93.4] | 89.0 [87.0, 91.1] | 92.4 [90.5, 93.8] | 93.3 [91.4, 94.9] | 91.5 [88.5, 93.9] | +| redux AND tuned ('redux', np.float64(0.1), np.float64(0.1)) | 96.0 [94.9, 97.1] | 88.5 [77.2, 92.2] | 88.5 [86.1, 90.7] | 91.6 [89.4, 93.2] | 94.6 [93.0, 96.0] | 88.9 [85.2, 91.7] | +| redux mean tuned ('redux', 0.75, np.float64(0.2)) | 96.5 [95.1, 97.6] | 90.4 [82.0, 93.4] | 88.9 [87.0, 90.9] | 92.4 [90.5, 93.8] | 93.4 [91.5, 95.1] | 91.4 [88.5, 93.8] | +| redux two-stage tuned ('redux', 0.1, 0.3, 32, 8, 30) | 96.8 [95.4, 98.0] | 90.8 [82.5, 93.9] | 89.2 [86.6, 91.7] | 92.7 [90.7, 94.2] | 92.8 [90.9, 94.5] | 92.6 [89.6, 94.9] | +| Silero thr + post tuned (0.25, 0.5, 0.1) 0.1 | 97.3 [95.8, 98.4] | 91.4 [82.3, 94.7] | 90.0 [87.5, 92.5] | 93.4 [91.2, 94.9] | 93.4 [91.6, 94.8] | 93.4 [90.0, 95.7] | +| ultra head thr, params + post tuned (0.25, 0.5, 0.2) ('ultra', 0.8) | 96.7 [94.9, 98.2] | 88.4 [78.7, 93.2] | 87.4 [86.1, 88.9] | 91.5 [89.4, 93.2] | 91.6 [89.4, 93.3] | 91.4 [88.3, 94.5] | +| ultra OR, params + post tuned (0.5, 0.5, 0.1) ('ultra', 0.1, 0.9) | 97.2 [95.7, 98.4] | 91.2 [81.0, 94.6] | 90.4 [88.6, 92.2] | 93.4 [91.3, 94.9] | 93.3 [91.5, 94.7] | 93.6 [90.1, 96.1] | +| ultra mean, params + post tuned (0.1, 0.5, 0.2) ('ultra', 0.5, 0.5) | 97.4 [96.0, 98.5] | 91.0 [81.8, 94.2] | 90.4 [88.1, 92.6] | 93.5 [91.4, 95.0] | 93.5 [91.9, 94.9] | 93.4 [90.1, 95.9] | +| ultra two-stage fusion, params + post tuned (0.5, 0.5, 0.03) ('ultra', 0.1, 0.3, 32, 0, 0) | 97.2 [95.7, 98.5] | 91.2 [81.4, 94.5] | 90.6 [88.2, 92.9] | 93.5 [91.4, 95.0] | 93.6 [91.8, 95.1] | 93.3 [90.0, 95.9] | +| ultra median gate, params + post tuned (0.25, 0.5, 0.1) ('ultra', 0.8) | 96.9 [95.0, 98.4] | 88.8 [79.0, 93.5] | 87.8 [86.6, 89.3] | 91.8 [89.7, 93.4] | 92.0 [89.9, 93.7] | 91.6 [88.4, 94.7] | +| ultra OR with gated head, params + post tuned (0.1, 0.5, 0.1) ('ultra', 0.92) | 97.4 [95.7, 98.7] | 89.7 [76.7, 94.5] | 90.7 [89.0, 92.2] | 93.2 [90.9, 94.9] | 93.9 [92.4, 95.2] | 92.5 [88.5, 95.6] | +| redux head thr, params + post tuned (0.5, 0.5, 0.1) ('redux', 0.6) | 97.3 [95.7, 98.5] | 92.2 [85.4, 94.7] | 89.3 [87.2, 91.4] | 93.3 [91.5, 94.7] | 93.0 [90.9, 94.8] | 93.6 [91.1, 95.5] | +| redux OR, params + post tuned (0.25, 0.5, 0.1) ('redux', 0.1, 0.9) | 97.4 [95.8, 98.5] | 92.6 [85.3, 95.2] | 90.6 [88.4, 92.9] | 93.9 [92.0, 95.3] | 92.9 [91.1, 94.5] | 94.8 [92.0, 96.9] | +| redux mean, params + post tuned (0.5, 0.5, 0.1) ('redux', 0.75, 0.2) | 97.4 [95.9, 98.5] | 92.6 [85.3, 95.3] | 90.8 [88.6, 93.1] | 94.0 [92.1, 95.3] | 93.3 [91.4, 94.9] | 94.7 [91.8, 96.8] | +| redux two-stage fusion, params + post tuned (0.5, 0.5, 0.03) ('redux', 0.1, 0.5, 32, 16, 30) | 97.4 [95.9, 98.6] | 92.1 [84.0, 95.2] | 90.7 [88.3, 93.1] | 93.8 [91.8, 95.3] | 93.6 [91.9, 95.1] | 94.0 [90.9, 96.3] | +| redux median gate, params + post tuned (0.25, 0.5, 0.03) ('redux', 0.8) | 97.6 [96.1, 98.7] | 92.0 [84.9, 94.6] | 90.8 [89.2, 92.6] | 93.9 [92.2, 95.2] | 94.8 [93.2, 96.0] | 93.1 [90.5, 95.2] | +| redux OR with gated head, params + post tuned (0.25, 0.5, 0.03) ('redux', 0.8) | 97.7 [96.4, 98.7] | 92.4 [85.1, 94.9] | 91.4 [89.7, 93.2] | 94.2 [92.5, 95.5] | 94.3 [92.7, 95.6] | 94.1 [91.4, 96.3] | + +## C. Paired differences in F1 points (bootstrap over recordings, 95% CI), pooled and per domain + +| comparison | split | vox | ami | ava | pooled | +|---|---|---|---|---|---| +| fusion defaults ultra minus ultra head .5 | all | 0.98 [0.61, 1.36] | -0.44 [-3.17, 0.99] | 4.50 [1.83, 7.48] | 1.51 [0.68, 2.32] | +| fusion defaults ultra minus Silero own defaults | all | 0.99 [0.72, 1.26] | 2.48 [1.65, 4.07] | 2.44 [1.90, 3.09] | 1.77 [1.45, 2.17] | +| ultra head .5 minus Silero own defaults | all | 0.01 [-0.42, 0.45] | 2.92 [0.68, 7.09] | -2.06 [-5.10, 0.80] | 0.26 [-0.72, 1.36] | +| gate .92 ultra minus ultra head .5 | all | -0.05 [-0.58, 0.55] | -4.29 [-6.73, -2.63] | 2.75 [0.05, 5.94] | -0.61 [-1.55, 0.39] | +| fusion defaults redux minus redux head .5 | all | 0.23 [0.11, 0.37] | -1.88 [-5.00, -0.18] | 2.14 [-0.17, 4.30] | 0.11 [-0.76, 0.86] | +| fusion defaults redux minus Silero own defaults | all | 1.18 [0.85, 1.51] | 3.61 [2.48, 5.58] | 2.78 [2.26, 3.40] | 2.30 [1.89, 2.78] | +| redux head .5 minus Silero own defaults | all | 0.95 [0.54, 1.33] | 5.49 [2.68, 10.52] | 0.63 [-1.67, 3.32] | 2.19 [1.14, 3.45] | +| gate .92 redux minus redux head .5 | all | -0.48 [-0.93, -0.10] | -1.41 [-2.74, -0.67] | 0.89 [-1.86, 3.46] | -0.32 [-0.98, 0.31] | +| fusion tuned ultra (thr/params only, default post) minus ultra head tuned (thr only) | held | 1.40 [0.54, 2.81] | 1.94 [1.10, 2.14] | 5.41 [2.44, 8.15] | 3.02 [1.88, 4.21] | +| fusion tuned ultra (thr/params only, default post) minus Silero thr tuned (default post) | held | 0.21 [-0.06, 0.56] | -0.41 [-0.69, 0.60] | 0.97 [0.50, 1.42] | 0.32 [0.07, 0.65] | +| fusion tuned redux (thr/params only, default post) minus redux head tuned (thr only) | held | 0.51 [0.25, 0.81] | 1.36 [0.55, 1.54] | 2.07 [-0.40, 4.60] | 1.26 [0.41, 2.14] | +| fusion tuned redux (thr/params only, default post) minus Silero thr tuned (default post) | held | 0.41 [0.20, 0.71] | 0.42 [0.06, 1.31] | 1.01 [0.67, 1.37] | 0.62 [0.41, 0.87] | + +### C2. Paired differences, everything tuned jointly with post-processing (held-out split) + +| comparison | split | vox | ami | ava | pooled | +|---|---|---|---|---|---| +| ultra fusion (0.5, 0.5, 0.03) minus best single detector tuned on tune (Silero) | held | -0.07 [-0.25, 0.10] | -0.29 [-0.82, 0.24] | 0.58 [0.37, 0.79] | 0.10 [-0.03, 0.26] | +| ultra gated head (0.25, 0.5, 0.1) minus best single detector tuned on tune (Silero) | held | -0.40 [-0.79, 0.01] | -2.69 [-3.28, -0.20] | -2.17 [-4.11, 0.07] | -1.58 [-2.38, -0.69] | +| ultra OR with gated head (0.1, 0.5, 0.1) minus best single detector tuned on tune (Silero) | held | 0.07 [-0.14, 0.28] | -1.71 [-5.53, 0.84] | 0.69 [-0.39, 1.73] | -0.15 [-0.73, 0.44] | +| ultra OR (0.5, 0.5, 0.1) minus best single detector tuned on tune (Silero) | held | -0.06 [-0.16, 0.06] | -0.20 [-1.31, 0.82] | 0.37 [-0.43, 1.16] | 0.04 [-0.25, 0.36] | +| ultra mean (0.1, 0.5, 0.2) minus best single detector tuned on tune (Silero) | held | 0.13 [-0.01, 0.29] | -0.42 [-0.53, 0.04] | 0.39 [-0.08, 0.97] | 0.09 [-0.11, 0.32] | +| ultra head tuned minus Silero tuned | held | -0.57 [-0.99, -0.14] | -3.03 [-3.63, -0.45] | -2.57 [-4.67, -0.12] | -1.87 [-2.75, -0.90] | +| ultra fusion minus Silero tuned | held | -0.07 [-0.25, 0.10] | -0.29 [-0.82, 0.24] | 0.58 [0.37, 0.79] | 0.10 [-0.03, 0.26] | +| redux fusion (0.5, 0.5, 0.03) minus best single detector tuned on tune (Silero) | held | 0.15 [0.06, 0.25] | 0.65 [-0.01, 1.68] | 0.71 [0.47, 0.99] | 0.46 [0.30, 0.66] | +| redux gated head (0.25, 0.5, 0.03) minus best single detector tuned on tune (Silero) | held | 0.27 [0.09, 0.62] | 0.60 [-0.16, 2.62] | 0.83 [-0.58, 2.24] | 0.54 [0.00, 1.17] | +| redux OR with gated head (0.25, 0.5, 0.03) minus best single detector tuned on tune (Silero) | held | 0.42 [0.23, 0.65] | 0.95 [0.19, 2.81] | 1.39 [0.28, 2.54] | 0.87 [0.43, 1.41] | +| redux OR (0.25, 0.5, 0.1) minus best single detector tuned on tune (Silero) | held | 0.07 [0.01, 0.12] | 1.13 [0.44, 3.00] | 0.64 [0.26, 1.04] | 0.50 [0.29, 0.83] | +| redux mean (0.5, 0.5, 0.1) minus best single detector tuned on tune (Silero) | held | 0.10 [0.02, 0.19] | 1.20 [0.61, 3.02] | 0.82 [0.20, 1.47] | 0.60 [0.31, 0.97] | +| redux head tuned minus Silero tuned | held | 0.03 [-0.10, 0.15] | 0.74 [-0.01, 3.13] | -0.70 [-2.95, 1.44] | -0.07 [-0.85, 0.74] | +| redux fusion minus Silero tuned | held | 0.15 [0.06, 0.25] | 0.65 [-0.01, 1.68] | 0.71 [0.47, 0.99] | 0.46 [0.30, 0.66] | + +## D. Leave-one-domain-out: parameters tuned on the other two domains (all their recordings), scored on the left-out domain (F1 and 95% CI) + +| system | vox (tuned on ami+ava) | ami (tuned on vox+ava) | ava (tuned on vox+ami) | mean | +|---|---|---|---|---| +| Silero thr | 97.1 [96.2, 97.9] | 91.9 [87.9, 94.2] | 85.4 [83.0, 87.8] | 91.5 | +| Silero all options | 97.4 [96.4, 98.2] | 92.7 [88.3, 95.2] | 88.0 [85.8, 90.2] | 92.7 | +| ultra head thr | 96.4 [95.1, 97.5] | 88.1 [84.1, 90.7] | 81.1 [76.8, 84.3] | 88.5 | +| ultra head thr + post | 97.0 [95.9, 97.9] | 88.5 [84.0, 91.4] | 82.6 [78.4, 85.7] | 89.4 | +| ultra OR | 97.2 [96.3, 97.9] | 91.9 [88.2, 94.1] | 86.3 [83.6, 88.3] | 91.8 | +| ultra OR + post | 97.6 [96.7, 98.3] | 93.1 [89.1, 95.4] | 88.1 [86.0, 89.9] | 92.9 | +| ultra mean | 97.2 [96.3, 98.0] | 90.9 [86.7, 93.5] | 82.8 [78.7, 85.7] | 90.3 | +| ultra mean + post | 97.6 [96.7, 98.3] | 92.2 [87.8, 94.8] | 86.4 [83.6, 88.6] | 92.1 | +| ultra two-stage | 97.3 [96.5, 98.1] | 91.7 [87.4, 94.3] | 88.1 [86.0, 90.2] | 92.4 | +| ultra two-stage + post | 97.6 [96.7, 98.3] | 93.0 [89.0, 95.4] | 88.9 [86.7, 90.9] | 93.2 | +| ultra gate + post | 97.1 [96.1, 98.0] | 89.2 [84.6, 92.3] | 82.9 [78.6, 85.9] | 89.7 | +| redux head thr | 97.2 [96.4, 98.0] | 91.5 [88.4, 93.5] | 84.8 [81.7, 87.2] | 91.2 | +| redux head thr + post | 97.6 [96.7, 98.4] | 93.1 [90.2, 94.9] | 86.8 [84.0, 89.2] | 92.5 | +| redux OR | 97.3 [96.5, 98.1] | 92.2 [88.8, 94.2] | 86.7 [84.5, 88.8] | 92.1 | +| redux OR + post | 97.6 [96.7, 98.4] | 93.6 [90.3, 95.5] | 88.5 [86.5, 90.4] | 93.2 | +| redux mean | 97.4 [96.6, 98.2] | 92.7 [89.6, 94.5] | 86.6 [84.4, 88.6] | 92.2 | +| redux mean + post | 97.7 [96.8, 98.4] | 93.8 [90.8, 95.6] | 89.6 [87.9, 91.3] | 93.7 | +| redux two-stage | 97.4 [96.6, 98.2] | 92.0 [87.6, 94.5] | 88.0 [85.9, 90.1] | 92.5 | +| redux two-stage + post | 97.7 [96.9, 98.4] | 93.1 [88.9, 95.6] | 89.1 [86.9, 91.2] | 93.3 | +| redux gate + post | 97.8 [96.9, 98.5] | 93.9 [91.2, 95.5] | 88.8 [86.9, 90.7] | 93.5 | + +Picks: Silero thr: [('sil', 0.05), ('sil', 0.075), ('sil', 0.02)]; Silero all options: [(0.25, 0.5, 0.2, ('sil', 0.1)), (0.25, 0.5, 0.1, ('sil', 0.2)), (0.25, 0.5, 0.1, ('sil', 0.05))]; ultra head thr: [('ultra', np.float64(0.3)), ('ultra', np.float64(0.65)), ('ultra', np.float64(0.2))]; ultra head thr + post: [(0.25, 0.5, 0.2, ('head', 'ultra', 0.7)), (0.25, 0.5, 0.2, ('head', 'ultra', 0.9)), (0.25, 0.5, 0.2, ('head', 'ultra', 0.4))]; ultra OR: [('ultra', np.float64(0.1), np.float64(0.9)), ('ultra', np.float64(0.1), np.float64(0.9)), ('ultra', np.float64(0.1), np.float64(0.8))]; ultra OR + post: [(0.5, 0.5, 0.1, ('or', 'ultra', 0.1, 0.9)), (0.5, 0.5, 0.03, ('or', 'ultra', 0.1, 0.9)), (0.25, 0.5, 0.2, ('or', 'ultra', 0.1, 0.9))]; ultra mean: [('ultra', 0.75, np.float64(0.25)), ('ultra', 0.75, np.float64(0.25)), ('ultra', 0.75, np.float64(0.1))]; ultra mean + post: [(0.25, 0.5, 0.2, ('mean', 'ultra', 0.75, 0.3)), (0.25, 0.5, 0.1, ('mean', 'ultra', 0.5, 0.5)), (0.25, 0.5, 0.2, ('mean', 'ultra', 0.75, 0.2))]; ultra two-stage: [('ultra', 0.1, 0.3, 32, 8, 0), ('ultra', 0.2, 0.3, 24, 8, 30), ('ultra', 0.1, 0.3, 32, 16, 30)]; ultra two-stage + post: [(0.25, 0.5, 0.1, ('two', 'ultra', 0.1, 0.9, 32, 16, 60)), (0.5, 0.5, 0.0, ('two', 'ultra', 0.1, 0.3, 16, 0, 0)), (0.25, 0.5, 0.2, ('two', 'ultra', 0.1, 0.9, 32, 32, 0))]; ultra gate + post: [(0.25, 0.5, 0.2, ('gate', 'ultra', 0.8)), (0.25, 0.5, 0.1, ('gate', 'ultra', 0.9)), (0.1, 0.5, 0.2, ('gate', 'ultra', 0.5))]; redux head thr: [('redux', np.float64(0.6)), ('redux', np.float64(0.65)), ('redux', np.float64(0.45))]; redux head thr + post: [(0.25, 0.5, 0.2, ('head', 'redux', 0.8)), (0.25, 0.5, 0.1, ('head', 'redux', 0.8)), (0.25, 0.5, 0.03, ('head', 'redux', 0.5))]; redux OR: [('redux', np.float64(0.1), np.float64(0.8)), ('redux', np.float64(0.1), np.float64(0.9)), ('redux', np.float64(0.1), np.float64(0.6))]; redux OR + post: [(0.25, 0.5, 0.1, ('or', 'redux', 0.1, 0.9)), (0.25, 0.5, 0.1, ('or', 'redux', 0.3, 0.9)), (0.25, 0.5, 0.1, ('or', 'redux', 0.1, 0.7))]; redux mean: [('redux', 0.75, np.float64(0.2)), ('redux', 0.75, np.float64(0.2)), ('redux', 0.75, np.float64(0.15))]; redux mean + post: [(0.5, 0.5, 0.1, ('mean', 'redux', 0.75, 0.2)), (0.5, 0.5, 0.1, ('mean', 'redux', 0.5, 0.4)), (0.25, 0.5, 0.1, ('mean', 'redux', 0.75, 0.2))]; redux two-stage: [('redux', 0.1, 0.3, 32, 8, 0), ('redux', 0.3, 0.3, 32, 8, 60), ('redux', 0.1, 0.3, 32, 32, 60)]; redux two-stage + post: [(0.5, 0.5, 0.03, ('two', 'redux', 0.1, 0.5, 32, 16, 0)), (0.5, 0.5, 0.03, ('two', 'redux', 0.3, 0.5, 32, 16, 30)), (0.5, 0.5, 0.0, ('two', 'redux', 0.1, 0.3, 32, 32, 0))]; redux gate + post: [(0.25, 0.5, 0.1, ('gate', 'redux', 0.8)), (0.25, 0.5, 0.03, ('gate', 'redux', 0.8)), (0.1, 0.5, 0.2, ('gate', 'redux', 0.8))] + +LODO paired deltas in F1 points (left-out domain; the best single detector is chosen by its score on the two tuning domains; CI over recordings of the left-out domain): + +| comparison | vox | ami | ava | mean of three | +|---|---|---|---|---| +| ultra two-stage (+post) minus best single (+post) | 0.24 [0.14, 0.38] | 0.39 [0.18, 0.76] | 0.83 [0.47, 1.23] | 0.49 | +| ultra gate (+post) minus best single (+post) | -0.23 [-0.41, -0.04] | -3.43 [-4.83, -2.30] | -5.16 [-8.53, -2.28] | -2.94 | +| ultra OR (+post) minus best single (+post) | 0.21 [0.07, 0.40] | 0.39 [0.11, 0.95] | 0.08 [-0.97, 1.13] | 0.23 | +| ultra mean (+post) minus best single (+post) | 0.24 [0.15, 0.36] | -0.50 [-0.82, -0.18] | -1.58 [-3.45, 0.12] | -0.61 | +| ultra head (+post) minus Silero (+post) | -0.36 [-0.63, -0.09] | -4.21 [-5.43, -3.11] | -5.43 [-8.82, -2.49] | -3.33 | +| redux two-stage (+post) minus best single (+post) | 0.35 [0.21, 0.55] | 0.43 [0.31, 0.60] | 2.30 [0.06, 4.49] | 1.03 | +| redux gate (+post) minus best single (+post) | 0.42 [0.23, 0.69] | 1.24 [0.26, 2.99] | 1.96 [0.65, 3.55] | 1.20 | +| redux OR (+post) minus best single (+post) | 0.25 [0.15, 0.39] | 0.93 [0.27, 2.12] | 1.65 [0.59, 2.75] | 0.94 | +| redux mean (+post) minus best single (+post) | 0.34 [0.20, 0.52] | 1.16 [0.38, 2.54] | 2.73 [1.25, 4.36] | 1.41 | +| redux head (+post) minus Silero (+post) | 0.25 [0.09, 0.48] | 0.44 [-0.39, 1.93] | -1.18 [-3.46, 1.17] | -0.16 | + +## E. Median-probability gate sweep (head thr .5, keep a run if its median probability >= m; unified post), all recordings + +| head | m | vox F1 | ami F1 | ava F1 | pooled F1 | music FA s/h | noise FA s/h | esc FA s/h | +|---|---|---|---|---|---|---|---|---| +| ultra | 0.5 | 96.4 [95.2, 97.5] | 89.7 [86.0, 92.0] | 82.7 [78.8, 85.2] | 90.8 [89.3, 92.0] | 2823 [2450, 3139] | 1979 [1485, 2558] | 1605 [1529, 1699] | +| ultra | 0.6 | 96.5 [95.3, 97.5] | 89.5 [85.8, 91.9] | 83.0 [79.3, 85.5] | 90.9 [89.4, 92.1] | 2744 [2355, 3084] | 1823 [1307, 2435] | 1476 [1387, 1581] | +| ultra | 0.7 | 96.6 [95.5, 97.5] | 89.1 [85.3, 91.6] | 83.9 [80.6, 86.1] | 91.1 [89.6, 92.2] | 2605 [2161, 2991] | 1582 [1062, 2204] | 1278 [1198, 1373] | +| ultra | 0.8 | 96.7 [95.7, 97.6] | 88.4 [84.1, 91.2] | 84.9 [82.2, 87.0] | 91.2 [89.7, 92.4] | 2389 [1905, 2848] | 1226 [774, 1826] | 959 [888, 1050] | +| ultra | 0.85 | 96.7 [95.7, 97.6] | 87.7 [83.1, 90.8] | 85.5 [83.2, 87.5] | 91.1 [89.6, 92.4] | 2144 [1613, 2682] | 948 [533, 1532] | 757 [693, 845] | +| ultra | 0.9 | 96.5 [95.4, 97.6] | 86.3 [81.0, 90.0] | 85.7 [83.5, 87.6] | 90.7 [89.0, 92.1] | 1831 [1245, 2420] | 576 [246, 1108] | 488 [440, 599] | +| ultra | 0.92 | 96.4 [95.2, 97.4] | 85.4 [79.6, 89.3] | 85.4 [83.2, 87.5] | 90.2 [88.4, 91.7] | 1625 [1074, 2194] | 398 [123, 911] | 395 [352, 492] | +| ultra | 0.94 | 96.1 [94.8, 97.3] | 83.7 [76.8, 88.4] | 84.5 [82.1, 86.9] | 89.4 [87.4, 91.2] | 1323 [788, 1872] | 77 [35, 125] | 285 [250, 371] | +| ultra | 0.96 | 95.6 [94.1, 97.0] | 80.5 [71.7, 86.5] | 82.2 [79.0, 85.3] | 87.7 [85.1, 89.9] | 761 [379, 1242] | 31 [14, 49] | 178 [148, 264] | +| ultra | 0.98 | 94.2 [91.7, 96.3] | 72.6 [58.6, 81.8] | 76.5 [71.8, 80.8] | 83.4 [79.3, 86.7] | 339 [23, 800] | 9 [2, 22] | 80 [58, 112] | +| ultra | 0.99 | 91.9 [88.6, 94.8] | 59.5 [39.0, 72.5] | 68.7 [62.0, 74.9] | 77.0 [71.2, 81.5] | 139 [6, 336] | 1 [0, 2] | 37 [18, 81] | +| redux | 0.5 | 97.3 [96.5, 98.1] | 92.2 [89.2, 94.1] | 85.4 [82.7, 87.6] | 92.8 [91.6, 93.7] | 2123 [1692, 2525] | 1599 [1187, 2094] | 1236 [1185, 1332] | +| redux | 0.6 | 97.4 [96.6, 98.2] | 92.3 [89.4, 94.1] | 86.1 [83.8, 88.2] | 93.0 [91.9, 93.9] | 1948 [1504, 2364] | 1404 [974, 1924] | 1055 [992, 1158] | +| redux | 0.7 | 97.4 [96.7, 98.2] | 92.3 [89.4, 94.0] | 87.2 [85.1, 89.2] | 93.3 [92.2, 94.2] | 1659 [1213, 2103] | 1095 [630, 1658] | 733 [701, 795] | +| redux | 0.8 | 97.4 [96.6, 98.1] | 92.1 [89.1, 94.0] | 88.0 [86.0, 89.8] | 93.5 [92.4, 94.4] | 1305 [864, 1776] | 639 [227, 1122] | 377 [325, 446] | +| redux | 0.85 | 97.3 [96.4, 98.1] | 91.9 [88.7, 93.8] | 87.8 [85.5, 89.8] | 93.3 [92.2, 94.3] | 1082 [646, 1567] | 297 [51, 613] | 259 [202, 328] | +| redux | 0.9 | 97.0 [96.1, 97.9] | 91.2 [87.4, 93.5] | 87.0 [84.3, 89.1] | 92.8 [91.5, 93.9] | 762 [378, 1229] | 79 [26, 139] | 167 [129, 213] | +| redux | 0.92 | 96.9 [95.8, 97.8] | 90.8 [86.6, 93.3] | 86.2 [83.2, 88.6] | 92.5 [91.0, 93.7] | 644 [308, 1055] | 23 [14, 31] | 131 [99, 179] | +| redux | 0.94 | 96.6 [95.3, 97.7] | 90.2 [85.5, 92.9] | 85.2 [81.8, 87.9] | 91.9 [90.3, 93.3] | 447 [176, 787] | 10 [2, 18] | 63 [50, 90] | +| redux | 0.96 | 96.3 [94.9, 97.5] | 88.9 [83.4, 92.1] | 83.6 [80.1, 86.6] | 91.0 [89.3, 92.6] | 265 [99, 474] | 4 [1, 7] | 34 [23, 60] | +| redux | 0.98 | 95.1 [92.9, 96.8] | 85.4 [77.3, 90.3] | 80.2 [76.0, 84.1] | 88.6 [86.1, 90.9] | 107 [13, 223] | 1 [0, 3] | 15 [9, 25] | +| redux | 0.99 | 93.3 [90.2, 95.6] | 79.8 [67.7, 87.1] | 75.4 [69.7, 80.3] | 85.0 [81.4, 88.1] | 39 [7, 93] | 0 [0, 0] | 8 [5, 11] | + +## F. Non-speech audio: seconds called speech per hour of audio and speech regions per hour (95% CI over files; domains: music 16 files, noise 6, esc 5) + +| system | music s/h | noise s/h | esc s/h | pooled s/h | pooled regions/h | +|---|---|---|---|---|---| +| Silero own defaults | 135 [0, 337] | 1 [0, 1] | 1 [0, 2] | 48 [0, 136] | 27 [1, 74] | +| Silero thr .5 unified | 135 [0, 335] | 0 [0, 1] | 2 [0, 5] | 48 [1, 135] | 27 [1, 74] | +| Silero tuned thr 0.05 | 287 [9, 698] | 13 [0, 26] | 14 [7, 24] | 109 [10, 290] | 46 [22, 81] | +| Silero all options tuned | 258 [6, 619] | 9 [0, 19] | 8 [2, 16] | 96 [7, 260] | 21 [6, 47] | +| ultra head .5 | 2823 [2450, 3139] | 1979 [1485, 2558] | 1605 [1529, 1699] | 2174 [1882, 2524] | 794 [646, 934] | +| ultra head tuned 0.35 | 3012 [2704, 3273] | 2291 [1855, 2799] | 1925 [1856, 2010] | 2446 [2184, 2746] | 714 [549, 864] | +| OR(S .5,ultra .5) | 2827 [2451, 3143] | 1979 [1485, 2558] | 1605 [1529, 1700] | 2176 [1883, 2526] | 791 [642, 932] | +| AND(S .5,ultra .5) | 131 [0, 322] | 0 [0, 1] | 1 [0, 4] | 46 [1, 130] | 30 [1, 79] | +| fusion defaults ultra | 165 [2, 402] | 2 [0, 4] | 5 [1, 10] | 60 [3, 163] | 24 [5, 56] | +| fusion tuned ultra (0.1, 0.3, 32, 8, 30) | 281 [12, 668] | 15 [0, 31] | 26 [18, 39] | 111 [15, 285] | 40 [20, 69] | +| ultra gate .92 | 1625 [1074, 2194] | 398 [123, 911] | 395 [352, 492] | 826 [523, 1229] | 158 [117, 207] | +| fusion with gated head ultra | 159 [1, 390] | 1 [0, 3] | 5 [1, 9] | 57 [2, 158] | 23 [4, 56] | +| OR(S .5, gated ultra) | 1648 [1087, 2230] | 399 [123, 911] | 396 [352, 493] | 835 [528, 1240] | 162 [120, 212] | +| redux head .5 | 2123 [1692, 2525] | 1599 [1187, 2094] | 1236 [1185, 1332] | 1685 [1435, 1987] | 1013 [904, 1147] | +| redux head tuned 0.6 | 1825 [1395, 2234] | 1299 [887, 1799] | 953 [898, 1044] | 1391 [1136, 1690] | 1028 [926, 1153] | +| OR(S .5,redux .5) | 2125 [1693, 2529] | 1599 [1187, 2094] | 1236 [1185, 1332] | 1686 [1435, 1989] | 1012 [902, 1146] | +| AND(S .5,redux .5) | 132 [0, 327] | 0 [0, 1] | 2 [0, 5] | 47 [1, 132] | 29 [1, 77] | +| fusion defaults redux | 164 [1, 402] | 2 [0, 5] | 5 [1, 9] | 59 [2, 164] | 23 [4, 56] | +| fusion tuned redux (0.1, 0.3, 32, 8, 30) | 277 [10, 661] | 14 [0, 31] | 24 [15, 36] | 108 [13, 282] | 42 [20, 74] | +| redux gate .92 | 644 [308, 1055] | 23 [14, 31] | 131 [99, 179] | 269 [122, 482] | 85 [50, 135] | +| fusion with gated head redux | 157 [1, 386] | 1 [0, 4] | 4 [0, 9] | 57 [2, 157] | 21 [2, 53] | +| OR(S .5, gated redux) | 660 [316, 1085] | 23 [14, 31] | 131 [99, 179] | 275 [124, 493] | 89 [51, 142] | +| Silero thr+post tuned (0.25, 0.5, 0.1) thr 0.1 | 255 [5, 614] | 9 [0, 19] | 8 [2, 16] | 95 [6, 258] | 21 [6, 50] | +| ultra head thr thr/params+post tuned (0.25, 0.5, 0.2) | 2462 [2002, 2891] | 1323 [861, 1934] | 1080 [1016, 1167] | 1656 [1347, 2052] | 485 [407, 572] | +| ultra fusion thr/params+post tuned (0.5, 0.5, 0.03) | 270 [6, 649] | 11 [0, 23] | 11 [2, 19] | 102 [8, 274] | 17 [6, 37] | +| ultra median gate thr/params+post tuned (0.25, 0.5, 0.1) | 2490 [2022, 2938] | 1274 [817, 1878] | 1060 [991, 1150] | 1642 [1325, 2055] | 302 [239, 368] | +| ultra OR with gated head thr/params+post tuned (0.1, 0.5, 0.1) | 1698 [1124, 2283] | 411 [132, 924] | 444 [397, 549] | 870 [560, 1278] | 140 [106, 179] | +| redux head thr thr/params+post tuned (0.5, 0.5, 0.1) | 1752 [1284, 2204] | 1236 [800, 1766] | 862 [812, 956] | 1317 [1044, 1636] | 382 [329, 442] | +| redux fusion thr/params+post tuned (0.5, 0.5, 0.03) | 272 [7, 653] | 10 [0, 22] | 18 [7, 30] | 104 [10, 276] | 22 [8, 47] | +| redux median gate thr/params+post tuned (0.25, 0.5, 0.03) | 1351 [900, 1831] | 659 [235, 1158] | 397 [343, 470] | 831 [571, 1153] | 218 [169, 280] | +| redux OR with gated head thr/params+post tuned (0.25, 0.5, 0.03) | 1354 [902, 1836] | 659 [235, 1158] | 397 [343, 470] | 832 [571, 1154] | 218 [169, 279] | diff --git a/scripts/vad_bench/real_recordings/results/fusion_grid.md b/scripts/vad_bench/real_recordings/results/fusion_grid.md new file mode 100644 index 0000000..f745653 --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/fusion_grid.md @@ -0,0 +1,26 @@ +# Fusion rule, ts = th = .5, default post, all recordings: delta F1 (pp) versus Silero .5 with the same post, paired bootstrap over recordings; FA = seconds called speech per hour on music/noise/esc pooled + +| head | pre cells (10 ms) | post cells | fill cells | pooled F1 | dF1 vs Silero .5 [95% CI] | non-speech FA s/h | +|---|---|---|---|---|---|---| +| ultra | 0 | 0 | 0 | 90.4 [88.2, 92.1] | 0.00 [0.00, 0.00] | 48 | +| ultra | 16 | 0 | 0 | 91.6 [89.6, 93.2] | 1.22 [1.04, 1.46] | 55 | +| ultra | 32 | 0 | 0 | 91.7 [89.8, 93.3] | 1.35 [1.13, 1.65] | 59 | +| ultra | 0 | 16 | 0 | 91.4 [89.4, 93.0] | 1.05 [0.84, 1.30] | 55 | +| ultra | 0 | 32 | 0 | 91.5 [89.6, 93.1] | 1.13 [0.88, 1.42] | 59 | +| ultra | 0 | 0 | 30 | 90.8 [88.7, 92.5] | 0.45 [0.37, 0.54] | 50 | +| ultra | 0 | 0 | 60 | 91.3 [89.2, 92.9] | 0.87 [0.73, 1.05] | 53 | +| ultra | 16 | 16 | 30 | 92.4 [90.5, 93.8] | 1.96 [1.65, 2.35] | 60 | +| ultra | 32 | 16 | 30 | 92.5 [90.6, 93.9] | 2.06 [1.71, 2.49] | 62 | +| ultra | 32 | 32 | 60 | 92.5 [90.7, 93.9] | 2.09 [1.72, 2.57] | 65 | +| ultra | 16 | 16 | 60 | 92.4 [90.6, 93.9] | 2.01 [1.68, 2.43] | 61 | +| redux | 0 | 0 | 0 | 90.4 [88.2, 92.1] | 0.00 [0.00, 0.00] | 48 | +| redux | 16 | 0 | 0 | 92.0 [90.0, 93.5] | 1.58 [1.35, 1.87] | 55 | +| redux | 32 | 0 | 0 | 92.2 [90.3, 93.6] | 1.76 [1.48, 2.13] | 58 | +| redux | 0 | 16 | 0 | 91.8 [89.8, 93.3] | 1.38 [1.11, 1.68] | 55 | +| redux | 0 | 32 | 0 | 91.9 [90.1, 93.4] | 1.54 [1.21, 1.92] | 58 | +| redux | 0 | 0 | 30 | 91.0 [88.9, 92.7] | 0.64 [0.56, 0.74] | 50 | +| redux | 0 | 0 | 60 | 91.7 [89.7, 93.3] | 1.36 [1.15, 1.59] | 54 | +| redux | 16 | 16 | 30 | 92.9 [91.1, 94.3] | 2.50 [2.09, 2.97] | 59 | +| redux | 32 | 16 | 30 | 93.0 [91.3, 94.4] | 2.62 [2.18, 3.15] | 62 | +| redux | 32 | 32 | 60 | 93.1 [91.5, 94.4] | 2.72 [2.23, 3.32] | 65 | +| redux | 16 | 16 | 60 | 93.0 [91.2, 94.4] | 2.57 [2.14, 3.07] | 61 | diff --git a/scripts/vad_bench/real_recordings/results/options.md b/scripts/vad_bench/real_recordings/results/options.md new file mode 100644 index 0000000..ca7a197 --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/options.md @@ -0,0 +1,43 @@ +# Option sweeps, all recordings, one option changed from the Silero defaults (thr .5, min speech .25 s, min pause/silence .1 s, pad .03 s) + +| option | value | vox F1 | ami F1 | ava F1 | pooled F1 | pooled P | pooled R | non-speech FA s/h (music / noise / esc) | +|---|---|---|---|---|---|---|---|---| +| (defaults) | - | 96.4 [95.4, 97.3] | 86.7 [78.9, 91.3] | 84.7 [81.6, 87.3] | 90.6 [88.4, 92.3] | 97.8 [97.3, 98.2] | 84.4 [80.7, 87.4] | 135 / 1 / 1 | +| threshold | 0.1 | 97.1 [96.3, 97.9] | 91.7 [87.5, 94.1] | 87.7 [85.2, 89.8] | 93.1 [91.8, 94.2] | 95.4 [94.6, 96.2] | 90.9 [88.7, 92.8] | 247 / 8 / 7 | +| threshold | 0.2 | 96.9 [96.1, 97.7] | 90.2 [85.1, 93.2] | 86.9 [84.2, 89.1] | 92.4 [90.9, 93.7] | 96.6 [95.9, 97.2] | 88.6 [85.9, 90.8] | 199 / 5 / 4 | +| threshold | 0.3 | 96.7 [95.9, 97.6] | 89.0 [83.0, 92.6] | 86.1 [83.3, 88.5] | 91.8 [90.0, 93.2] | 97.1 [96.5, 97.7] | 87.0 [84.0, 89.5] | 168 / 2 / 2 | +| threshold | 0.4 | 96.6 [95.6, 97.4] | 87.9 [81.0, 92.0] | 85.4 [82.5, 87.9] | 91.2 [89.2, 92.8] | 97.5 [97.0, 98.0] | 85.7 [82.3, 88.5] | 152 / 1 / 1 | +| threshold | 0.6 | 96.2 [95.1, 97.1] | 85.5 [76.6, 90.7] | 83.9 [80.5, 86.7] | 89.9 [87.5, 91.9] | 98.0 [97.6, 98.4] | 83.0 [79.0, 86.4] | 116 / 1 / 0 | +| threshold | 0.7 | 95.9 [94.8, 96.9] | 84.1 [74.0, 89.9] | 82.9 [79.2, 85.9] | 89.2 [86.5, 91.3] | 98.3 [97.8, 98.6] | 81.6 [77.2, 85.3] | 96 / 0 / 0 | +| min speech s | 0.1 | 96.4 [95.4, 97.3] | 87.1 [79.7, 91.5] | 84.8 [81.8, 87.3] | 90.7 [88.6, 92.4] | 97.7 [97.2, 98.1] | 84.7 [81.1, 87.7] | 138 / 1 / 2 | +| min speech s | 0.5 | 96.2 [95.1, 97.1] | 85.3 [76.6, 90.4] | 83.6 [80.0, 86.5] | 89.8 [87.5, 91.7] | 98.0 [97.6, 98.5] | 82.9 [78.9, 86.2] | 130 / 0 / 0 | +| min silence s (pause) | 0.2 | 96.8 [95.8, 97.7] | 87.1 [79.4, 91.7] | 85.0 [81.9, 87.6] | 91.0 [88.8, 92.7] | 97.8 [97.3, 98.2] | 85.0 [81.3, 88.1] | 136 / 1 / 1 | +| min silence s (pause) | 0.5 | 97.8 [97.0, 98.5] | 89.0 [81.4, 93.3] | 86.9 [83.9, 89.4] | 92.4 [90.3, 94.1] | 97.5 [97.0, 98.0] | 87.9 [84.0, 91.0] | 141 / 1 / 1 | +| speech pad s | 0.0 | 95.8 [94.7, 96.8] | 85.5 [77.4, 90.3] | 83.6 [80.3, 86.3] | 89.7 [87.4, 91.5] | 98.1 [97.6, 98.5] | 82.6 [78.9, 85.7] | 131 / 0 / 1 | +| speech pad s | 0.1 | 97.4 [96.5, 98.2] | 89.0 [81.8, 93.1] | 87.0 [84.1, 89.3] | 92.3 [90.2, 93.9] | 96.9 [96.3, 97.4] | 88.0 [84.4, 91.0] | 145 / 1 / 1 | +| speech pad s | 0.2 | 97.7 [96.7, 98.4] | 90.9 [84.3, 94.6] | 88.4 [85.8, 90.6] | 93.3 [91.4, 94.8] | 95.5 [94.8, 96.2] | 91.1 [87.7, 93.9] | 156 / 1 / 1 | + +## Silero threshold with the unified post (ms .1, pause .2, pad 0), thresholds below .1 + +| thr | vox F1 | ami F1 | ava F1 | pooled F1 | nonspeech FA s/h | +|---|---|---|---|---|---| +| 0.02 | 96.5 [95.2, 97.5] | 93.8 [91.4, 95.3] | 85.4 [83.0, 87.8] | 93.0 [91.8, 94.0] | 338 / 22 / 28 | +| 0.05 | 97.1 [96.2, 97.9] | 92.6 [89.1, 94.7] | 87.1 [84.8, 89.3] | 93.3 [92.0, 94.3] | 287 / 13 / 14 | +| 0.075 | 97.2 [96.3, 97.9] | 91.9 [87.9, 94.2] | 87.3 [84.9, 89.6] | 93.1 [91.8, 94.2] | 262 / 10 / 10 | +| 0.1 | 97.1 [96.3, 97.9] | 91.4 [87.1, 93.8] | 87.3 [84.8, 89.4] | 92.9 [91.5, 94.1] | 247 / 9 / 9 | +| 0.2 | 97.0 [96.1, 97.8] | 89.9 [84.7, 93.0] | 86.4 [83.7, 88.7] | 92.2 [90.6, 93.5] | 197 / 5 / 5 | + +## Head thresholds (unified post: bridge .1, min speech .1, pause .2, pad 0), all recordings + +| head | thr | vox F1 | ami F1 | ava F1 | pooled F1 | pooled P | pooled R | non-speech FA s/h (music / noise / esc) | +|---|---|---|---|---|---|---|---|---| +| ultra | 0.3 | 96.4 [95.1, 97.5] | 91.0 [87.6, 93.2] | 81.8 [77.5, 84.7] | 91.0 [89.4, 92.2] | 88.9 [86.3, 90.8] | 93.1 [91.7, 94.4] | 3074 / 2400 / 2039 | +| ultra | 0.5 | 96.4 [95.2, 97.5] | 89.7 [86.0, 92.0] | 82.7 [78.8, 85.2] | 90.8 [89.3, 92.0] | 91.3 [89.1, 92.8] | 90.4 [88.7, 92.0] | 2823 / 1979 / 1605 | +| ultra | 0.7 | 96.1 [95.0, 97.1] | 87.4 [83.1, 90.2] | 83.2 [80.0, 85.3] | 90.2 [88.6, 91.4] | 93.6 [92.0, 94.8] | 87.0 [84.9, 88.9] | 2483 / 1501 / 1172 | +| ultra | 0.9 | 94.5 [93.2, 95.7] | 81.4 [75.4, 85.4] | 82.0 [79.8, 84.1] | 87.3 [85.5, 88.9] | 96.8 [96.0, 97.4] | 79.5 [76.8, 82.1] | 1677 / 665 / 550 | +| ultra | 0.95 | 92.7 [91.1, 94.2] | 76.4 [68.9, 81.5] | 79.3 [76.7, 81.9] | 84.4 [82.2, 86.4] | 98.0 [97.5, 98.4] | 74.1 [70.8, 77.1] | 1181 / 278 / 323 | +| redux | 0.3 | 97.1 [95.9, 98.0] | 92.1 [88.8, 94.2] | 82.8 [78.9, 85.9] | 91.9 [90.3, 93.1] | 88.9 [86.3, 90.8] | 95.2 [94.2, 96.0] | 2724 / 2307 / 1809 | +| redux | 0.5 | 97.3 [96.5, 98.1] | 92.2 [89.2, 94.1] | 85.4 [82.7, 87.6] | 92.8 [91.6, 93.7] | 92.8 [91.3, 94.1] | 92.7 [91.4, 93.8] | 2123 / 1599 / 1236 | +| redux | 0.7 | 97.0 [96.2, 97.8] | 91.2 [87.9, 93.2] | 86.2 [84.2, 88.1] | 92.6 [91.4, 93.5] | 95.6 [94.7, 96.4] | 89.7 [88.0, 91.2] | 1509 / 962 / 667 | +| redux | 0.9 | 95.2 [94.1, 96.3] | 87.8 [83.3, 90.5] | 83.6 [81.0, 85.9] | 90.1 [88.6, 91.4] | 97.9 [97.4, 98.3] | 83.5 [81.1, 85.6] | 779 / 243 / 188 | +| redux | 0.95 | 93.5 [92.0, 94.9] | 84.8 [79.2, 88.1] | 80.7 [77.5, 83.6] | 87.8 [85.9, 89.4] | 98.4 [98.1, 98.7] | 79.2 [76.3, 81.7] | 496 / 90 / 102 | diff --git a/scripts/vad_bench/real_recordings/results/recordings.json b/scripts/vad_bench/real_recordings/results/recordings.json new file mode 100644 index 0000000..59a52ce --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/recordings.json @@ -0,0 +1,690 @@ +[ +{ +"id": "vox_dev_0004", +"domain": "vox", +"split": "tune", +"dur": 44.5126875, +"nonspeech": false, +"speech_ref_s": 32.6 +}, +{ +"id": "vox_dev_0012", +"domain": "vox", +"split": "tune", +"dur": 797.49225, +"nonspeech": false, +"speech_ref_s": 784.88 +}, +{ +"id": "vox_dev_0020", +"domain": "vox", +"split": "tune", +"dur": 145.344, +"nonspeech": false, +"speech_ref_s": 144.52 +}, +{ +"id": "vox_dev_0028", +"domain": "vox", +"split": "tune", +"dur": 759.379625, +"nonspeech": false, +"speech_ref_s": 730.0 +}, +{ +"id": "vox_dev_0036", +"domain": "vox", +"split": "tune", +"dur": 178.464, +"nonspeech": false, +"speech_ref_s": 157.44 +}, +{ +"id": "vox_dev_0044", +"domain": "vox", +"split": "tune", +"dur": 1032.384, +"nonspeech": false, +"speech_ref_s": 1016.48 +}, +{ +"id": "vox_dev_0052", +"domain": "vox", +"split": "tune", +"dur": 653.8449375, +"nonspeech": false, +"speech_ref_s": 637.76 +}, +{ +"id": "vox_dev_0060", +"domain": "vox", +"split": "tune", +"dur": 142.368, +"nonspeech": false, +"speech_ref_s": 138.68 +}, +{ +"id": "vox_dev_0068", +"domain": "vox", +"split": "tune", +"dur": 298.92, +"nonspeech": false, +"speech_ref_s": 215.96 +}, +{ +"id": "vox_dev_0076", +"domain": "vox", +"split": "tune", +"dur": 401.136, +"nonspeech": false, +"speech_ref_s": 238.16 +}, +{ +"id": "vox_dev_0084", +"domain": "vox", +"split": "tune", +"dur": 192.24, +"nonspeech": false, +"speech_ref_s": 141.44 +}, +{ +"id": "vox_dev_0092", +"domain": "vox", +"split": "tune", +"dur": 472.32, +"nonspeech": false, +"speech_ref_s": 460.64 +}, +{ +"id": "vox_dev_0100", +"domain": "vox", +"split": "tune", +"dur": 478.104, +"nonspeech": false, +"speech_ref_s": 402.92 +}, +{ +"id": "vox_dev_0108", +"domain": "vox", +"split": "tune", +"dur": 395.88, +"nonspeech": false, +"speech_ref_s": 386.76 +}, +{ +"id": "vox_dev_0116", +"domain": "vox", +"split": "tune", +"dur": 473.136, +"nonspeech": false, +"speech_ref_s": 379.36 +}, +{ +"id": "vox_dev_0124", +"domain": "vox", +"split": "tune", +"dur": 364.608, +"nonspeech": false, +"speech_ref_s": 224.36 +}, +{ +"id": "vox_dev_0132", +"domain": "vox", +"split": "tune", +"dur": 443.736, +"nonspeech": false, +"speech_ref_s": 395.0 +}, +{ +"id": "vox_dev_0140", +"domain": "vox", +"split": "tune", +"dur": 194.136, +"nonspeech": false, +"speech_ref_s": 193.08 +}, +{ +"id": "vox_dev_0148", +"domain": "vox", +"split": "tune", +"dur": 142.92, +"nonspeech": false, +"speech_ref_s": 125.76 +}, +{ +"id": "vox_dev_0156", +"domain": "vox", +"split": "tune", +"dur": 259.160875, +"nonspeech": false, +"speech_ref_s": 254.0 +}, +{ +"id": "vox_dev_0164", +"domain": "vox", +"split": "tune", +"dur": 70.3216875, +"nonspeech": false, +"speech_ref_s": 66.56 +}, +{ +"id": "vox_dev_0172", +"domain": "vox", +"split": "tune", +"dur": 587.04, +"nonspeech": false, +"speech_ref_s": 585.28 +}, +{ +"id": "vox_dev_0180", +"domain": "vox", +"split": "tune", +"dur": 259.944, +"nonspeech": false, +"speech_ref_s": 228.36 +}, +{ +"id": "vox_dev_0188", +"domain": "vox", +"split": "tune", +"dur": 439.416, +"nonspeech": false, +"speech_ref_s": 379.48 +}, +{ +"id": "vox_dev_0196", +"domain": "vox", +"split": "tune", +"dur": 138.24, +"nonspeech": false, +"speech_ref_s": 129.36 +}, +{ +"id": "vox_dev_0204", +"domain": "vox", +"split": "tune", +"dur": 207.552, +"nonspeech": false, +"speech_ref_s": 200.8 +}, +{ +"id": "vox_dev_0212", +"domain": "vox", +"split": "tune", +"dur": 175.92, +"nonspeech": false, +"speech_ref_s": 165.96 +}, +{ +"id": "vox_test_0004", +"domain": "vox", +"split": "held", +"dur": 1200.064, +"nonspeech": false, +"speech_ref_s": 1158.25 +}, +{ +"id": "vox_test_0012", +"domain": "vox", +"split": "held", +"dur": 411.648, +"nonspeech": false, +"speech_ref_s": 242.47 +}, +{ +"id": "vox_test_0020", +"domain": "vox", +"split": "held", +"dur": 1200.064, +"nonspeech": false, +"speech_ref_s": 1074.46 +}, +{ +"id": "vox_test_0028", +"domain": "vox", +"split": "held", +"dur": 386.752, +"nonspeech": false, +"speech_ref_s": 358.5 +}, +{ +"id": "vox_test_0036", +"domain": "vox", +"split": "held", +"dur": 43.904, +"nonspeech": false, +"speech_ref_s": 39.62 +}, +{ +"id": "vox_test_0044", +"domain": "vox", +"split": "held", +"dur": 1200.064, +"nonspeech": false, +"speech_ref_s": 1084.95 +}, +{ +"id": "vox_test_0052", +"domain": "vox", +"split": "held", +"dur": 262.592, +"nonspeech": false, +"speech_ref_s": 247.32 +}, +{ +"id": "vox_test_0060", +"domain": "vox", +"split": "held", +"dur": 329.024, +"nonspeech": false, +"speech_ref_s": 319.39 +}, +{ +"id": "vox_test_0068", +"domain": "vox", +"split": "held", +"dur": 1200.064, +"nonspeech": false, +"speech_ref_s": 1107.02 +}, +{ +"id": "ami_test_0001", +"domain": "ami", +"split": "held", +"dur": 1505.642625, +"nonspeech": false, +"speech_ref_s": 978.15 +}, +{ +"id": "ami_test_0003", +"domain": "ami", +"split": "held", +"dur": 2345.493375, +"nonspeech": false, +"speech_ref_s": 2014.9 +}, +{ +"id": "ami_test_0005", +"domain": "ami", +"split": "held", +"dur": 838.9973125, +"nonspeech": false, +"speech_ref_s": 605.13 +}, +{ +"id": "ami_validation_0001", +"domain": "ami", +"split": "tune", +"dur": 2417.0666875, +"nonspeech": false, +"speech_ref_s": 2129.77 +}, +{ +"id": "ami_validation_0004", +"domain": "ami", +"split": "tune", +"dur": 1616.064, +"nonspeech": false, +"speech_ref_s": 1333.0 +}, +{ +"id": "ami_validation_0007", +"domain": "ami", +"split": "tune", +"dur": 1113.845375, +"nonspeech": false, +"speech_ref_s": 815.5600000000001 +}, +{ +"id": "ami_validation_0010", +"domain": "ami", +"split": "tune", +"dur": 2393.0026875, +"nonspeech": false, +"speech_ref_s": 2201.42 +}, +{ +"id": "ami_validation_0013", +"domain": "ami", +"split": "tune", +"dur": 1546.410625, +"nonspeech": false, +"speech_ref_s": 1283.64 +}, +{ +"id": "ami_validation_0016", +"domain": "ami", +"split": "tune", +"dur": 1345.322625, +"nonspeech": false, +"speech_ref_s": 876.47 +}, +{ +"id": "ava_human_8aMv-ZGD4ic_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 727.57 +}, +{ +"id": "ava_human_9Y_l9NsnYE0_clip", +"domain": "ava", +"split": "tune", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 579.62 +}, +{ +"id": "ava_human_CrlfWnsS7ac_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 472.42 +}, +{ +"id": "ava_human_J4bt4y9ShTA_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 491.95 +}, +{ +"id": "ava_human_N5UD8FGzDek_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 587.04 +}, +{ +"id": "ava_human_NEQ7Wpf-EtI_clip", +"domain": "ava", +"split": "tune", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 516.22 +}, +{ +"id": "ava_human_OfMdakd4bHI_clip", +"domain": "ava", +"split": "tune", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 488.40000000000003 +}, +{ +"id": "ava_human_P90hF2S1JzA_clip", +"domain": "ava", +"split": "tune", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 628.86 +}, +{ +"id": "ava_human_QCLQYnt3aMo_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 553.0 +}, +{ +"id": "ava_human_UOyyTUX5Vo4_clip", +"domain": "ava", +"split": "tune", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 274.57 +}, +{ +"id": "ava_human_cKA-qeZuH_w_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 594.21 +}, +{ +"id": "ava_human_fNcxxBjEOgw_clip", +"domain": "ava", +"split": "tune", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 589.67 +}, +{ +"id": "ava_human_yn9WN9lsHRE_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 503.46000000000004 +}, +{ +"id": "ava_human_yo-Kg2YxlZs_clip", +"domain": "ava", +"split": "held", +"dur": 900.0, +"nonspeech": false, +"speech_ref_s": 637.28 +}, +{ +"id": "music_music_0020", +"domain": "music", +"split": "tune", +"dur": 220.9959375, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0060", +"domain": "music", +"split": "held", +"dur": 133.4335, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0100", +"domain": "music", +"split": "tune", +"dur": 48.065, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0140", +"domain": "music", +"split": "tune", +"dur": 200.5420625, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0180", +"domain": "music", +"split": "tune", +"dur": 208.896, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0220", +"domain": "music", +"split": "tune", +"dur": 154.2530625, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0260", +"domain": "music", +"split": "held", +"dur": 187.062875, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0300", +"domain": "music", +"split": "held", +"dur": 376.816, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0340", +"domain": "music", +"split": "held", +"dur": 238.027, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0380", +"domain": "music", +"split": "tune", +"dur": 115.252, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0420", +"domain": "music", +"split": "held", +"dur": 220.604, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0460", +"domain": "music", +"split": "tune", +"dur": 229.903, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0500", +"domain": "music", +"split": "tune", +"dur": 315.951, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0540", +"domain": "music", +"split": "held", +"dur": 222.3281875, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0580", +"domain": "music", +"split": "tune", +"dur": 204.0424375, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "music_music_0620", +"domain": "music", +"split": "held", +"dur": 203.8595625, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "noise_noise_00", +"domain": "noise", +"split": "held", +"dur": 613.7449375, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "noise_noise_01", +"domain": "noise", +"split": "tune", +"dur": 640.90175, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "noise_noise_02", +"domain": "noise", +"split": "tune", +"dur": 600.30125, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "noise_noise_03", +"domain": "noise", +"split": "held", +"dur": 603.4269375, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "noise_noise_04", +"domain": "noise", +"split": "held", +"dur": 879.5681875, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "noise_noise_05", +"domain": "noise", +"split": "tune", +"dur": 263.109375, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "esc_esc_00", +"domain": "esc", +"split": "held", +"dur": 600.0, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "esc_esc_01", +"domain": "esc", +"split": "tune", +"dur": 600.0, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "esc_esc_02", +"domain": "esc", +"split": "tune", +"dur": 600.0, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "esc_esc_03", +"domain": "esc", +"split": "tune", +"dur": 600.0, +"nonspeech": true, +"speech_ref_s": 0.0 +}, +{ +"id": "esc_esc_04", +"domain": "esc", +"split": "held", +"dur": 100.0, +"nonspeech": true, +"speech_ref_s": 0.0 +} +] \ No newline at end of file diff --git a/scripts/vad_bench/real_recordings/results/seg.md b/scripts/vad_bench/real_recordings/results/seg.md new file mode 100644 index 0000000..38b1932 --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/seg.md @@ -0,0 +1,143 @@ +# Segmentation: non-speech sent to the decoder (segment_by_vad replica, verified against the CLI), per hour of audio + +## S1. Seconds of reference non-speech inside the decoded segments, per hour of audio (95% CI over recordings), old behaviour (trim 0) -> default (trim 0.3). Last column pair: share of reference speech outside the segments (speech lost), trim 0.3 + +| system | vox trim0 | vox trim.3 | vox lost % | ami trim0 | ami trim.3 | ami lost % | ava trim0 | ava trim.3 | ava lost % | music trim0 | music trim.3 | noise trim0 | noise trim.3 | esc trim0 | esc trim.3 | +|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---| +| silero | 346 [233, 482] | 273 [183, 379] | 0.2 [0.0, 0.4] | 594 [453, 796] | 355 [303, 427] | 5.3 [1.6, 11.6] | 1273 [1128, 1415] | 885 [749, 1015] | 6.3 [3.7, 9.1] | 375 [0, 905] | 213 [0, 524] | 20 [0, 53] | 1 [0, 3] | 43 [0, 130] | 1 [0, 4] | +| silero_t03 | 346 [233, 482] | 275 [183, 384] | 0.1 [0.0, 0.3] | 617 [460, 841] | 390 [328, 478] | 3.4 [1.2, 7.0] | 1293 [1132, 1465] | 898 [769, 1022] | 5.6 [3.4, 8.1] | 414 [0, 981] | 248 [0, 604] | 78 [0, 151] | 4 [0, 8] | 48 [0, 145] | 3 [0, 9] | +| silero_t02 | 345 [232, 481] | 273 [183, 381] | 0.1 [0.0, 0.2] | 634 [473, 858] | 418 [349, 521] | 2.5 [0.9, 5.1] | 1299 [1135, 1475] | 894 [769, 1015] | 5.4 [3.1, 7.9] | 533 [105, 1118] | 288 [4, 696] | 160 [0, 328] | 8 [0, 17] | 135 [43, 188] | 6 [1, 11] | +| silero_t01 | 346 [233, 482] | 275 [183, 384] | 0.1 [0.0, 0.2] | 656 [487, 892] | 451 [371, 566] | 1.8 [0.7, 3.7] | 1306 [1143, 1466] | 924 [803, 1042] | 4.6 [2.4, 7.0] | 600 [183, 1163] | 311 [8, 752] | 189 [0, 359] | 11 [0, 23] | 221 [56, 352] | 11 [3, 19] | +| ultra | 346 [233, 482] | 317 [211, 445] | 0.0 [0.0, 0.1] | 687 [499, 949] | 598 [446, 806] | 0.6 [0.4, 1.0] | 1416 [1212, 1640] | 1348 [1148, 1574] | 0.3 [0.1, 0.5] | 3587 [3558, 3600] | 3419 [3326, 3490] | 3600 [3600, 3600] | 3029 [2841, 3236] | 3600 [3600, 3600] | 3378 [3334, 3416] | +| redux | 346 [233, 482] | 296 [192, 421] | 0.1 [0.0, 0.1] | 687 [499, 949] | 620 [453, 852] | 0.3 [0.2, 0.5] | 1416 [1212, 1640] | 1328 [1130, 1553] | 0.2 [0.1, 0.4] | 3600 [3600, 3600] | 3269 [3002, 3442] | 3598 [3594, 3600] | 2952 [2753, 3190] | 3600 [3600, 3600] | 3378 [3288, 3448] | +| ultra_t09 | 346 [233, 482] | 295 [194, 413] | 0.2 [0.1, 0.2] | 667 [492, 914] | 499 [384, 671] | 1.7 [1.0, 2.9] | 1416 [1212, 1640] | 1226 [1049, 1416] | 1.2 [0.5, 1.9] | 3600 [3600, 3600] | 3095 [2840, 3316] | 3086 [2787, 3409] | 1866 [1431, 2357] | 3600 [3600, 3600] | 2782 [2659, 2902] | +| redux_t09 | 345 [230, 481] | 266 [177, 370] | 0.2 [0.1, 0.3] | 683 [499, 945] | 542 [414, 723] | 1.0 [0.6, 1.6] | 1391 [1202, 1602] | 1133 [989, 1282] | 1.8 [0.7, 2.9] | 3125 [2591, 3527] | 2170 [1545, 2757] | 2409 [1931, 2866] | 1117 [584, 1739] | 3174 [2921, 3456] | 1418 [1176, 1562] | +| fusion_ultra | 346 [233, 482] | 260 [173, 367] | 0.1 [0.0, 0.2] | 637 [479, 857] | 368 [317, 438] | 4.1 [1.4, 8.7] | 1297 [1134, 1468] | 861 [720, 994] | 5.6 [3.1, 8.1] | 548 [150, 1077] | 263 [6, 639] | 80 [0, 158] | 4 [0, 8] | 217 [144, 304] | 11 [5, 16] | +| fusion_redux | 346 [233, 482] | 253 [170, 356] | 0.1 [0.0, 0.3] | 637 [479, 857] | 360 [311, 425] | 4.1 [1.3, 8.6] | 1289 [1133, 1444] | 869 [720, 1013] | 5.6 [3.0, 8.3] | 449 [36, 1019] | 261 [1, 636] | 111 [0, 231] | 5 [0, 11] | 173 [43, 290] | 9 [2, 16] | +| fusiontuned_ultra | 346 [233, 482] | 283 [187, 396] | 0.1 [0.0, 0.2] | 672 [491, 923] | 485 [385, 626] | 1.4 [0.5, 2.9] | 1318 [1158, 1483] | 949 [839, 1059] | 4.6 [2.4, 7.0] | 812 [391, 1399] | 343 [21, 809] | 358 [33, 704] | 30 [1, 68] | 906 [715, 1258] | 56 [40, 97] | +| fusiontuned_redux | 346 [233, 482] | 280 [185, 395] | 0.1 [0.0, 0.2] | 672 [491, 923] | 483 [384, 624] | 1.5 [0.6, 3.0] | 1325 [1165, 1485] | 937 [818, 1049] | 4.2 [2.1, 6.5] | 812 [392, 1396] | 342 [21, 810] | 360 [33, 705] | 33 [1, 75] | 891 [695, 1233] | 48 [34, 72] | +| gate_ultra | 346 [233, 482] | 258 [172, 362] | 0.9 [0.4, 1.5] | 637 [468, 881] | 412 [329, 546] | 3.9 [2.1, 6.7] | 1344 [1167, 1545] | 927 [802, 1050] | 5.1 [3.2, 7.2] | 3085 [2761, 3368] | 1913 [1316, 2490] | 1419 [1028, 1940] | 511 [173, 1060] | 3237 [2996, 3426] | 957 [858, 1157] | +| gate_redux | 345 [230, 481] | 238 [158, 336] | 0.9 [0.3, 1.7] | 660 [486, 907] | 426 [343, 551] | 2.1 [0.9, 4.1] | 1340 [1164, 1539] | 832 [708, 954] | 7.8 [4.3, 11.6] | 1927 [1317, 2544] | 979 [486, 1549] | 311 [203, 416] | 40 [21, 59] | 1802 [1637, 2078] | 200 [145, 290] | +| gate98_redux | 339 [228, 471] | 221 [146, 317] | 3.5 [1.2, 6.5] | 560 [425, 765] | 313 [267, 373] | 7.6 [3.3, 15.0] | 1254 [1089, 1431] | 708 [578, 844] | 16.5 [10.5, 22.7] | 797 [252, 1405] | 181 [27, 406] | 20 [0, 53] | 2 [0, 4] | 386 [288, 468] | 23 [15, 34] | +| orgate_ultra | 346 [233, 482] | 271 [179, 381] | 0.1 [0.0, 0.2] | 648 [478, 887] | 435 [358, 543] | 1.8 [0.6, 3.8] | 1344 [1167, 1545] | 969 [840, 1091] | 2.9 [1.7, 4.3] | 3085 [2761, 3368] | 1942 [1325, 2539] | 1437 [1040, 1949] | 512 [173, 1061] | 3237 [2996, 3426] | 958 [860, 1157] | +| orgate_redux | 346 [233, 482] | 254 [170, 357] | 0.1 [0.0, 0.2] | 663 [491, 908] | 445 [356, 572] | 1.5 [0.5, 3.1] | 1343 [1167, 1541] | 884 [755, 1001] | 4.7 [2.5, 7.2] | 1939 [1322, 2567] | 1015 [491, 1613] | 311 [203, 416] | 40 [21, 59] | 1839 [1637, 2134] | 201 [145, 290] | +| oracle | 344 [230, 481] | 230 [151, 324] | 0.0 [0.0, 0.0] | 638 [459, 894] | 375 [277, 523] | 0.0 [0.0, 0.0] | 1349 [1162, 1556] | 833 [700, 962] | 0.0 [0.0, 0.0] | 0 [0, 0] | 0 [0, 0] | 0 [0, 0] | 0 [0, 0] | 0 [0, 0] | 0 [0, 0] | + +Reading: for music/noise/esc every decoded second is non-speech; 3600 means the whole hour was decoded. Short files (<= 30 s) are never cut or trimmed by the segmenter (not present here: all files are >= 6 min). + +## S2. Cut policy (trim 0.3): non-speech seconds decoded per hour [95% CI], speech lost %, number of segments per hour, hard cuts per hour and cuts that land in reference speech per hour + +policies: base = shipped rule; soft = shipped rule, but before a hard cut take the midpoint of the longest silent gap of any length in the window; longest = cut at the longest pause (>= min pause) in the window; gap5 / gap2 = shipped rule plus a cut at every pause of at least 5 s / 2 s (decoded segments can then be shorter than 1 s apart). + +### vox (4.44 h) + +| detector | policy | non-speech s/h | speech lost % | segments/h | hard cuts/h | cuts in ref speech/h (all) | hard cuts in ref speech/h | +|---|---|---|---|---|---|---|---| +| silero | base | 273 [183, 379] | 0.2 [0.0, 0.4] | 142 | 9.7 | 101.8 | 9.2 | +| silero | soft | 273 [183, 379] | 0.2 [0.0, 0.4] | 142 | 9.0 | 102.0 | 9.0 | +| silero | longest | 173 [116, 247] | 0.5 [0.1, 1.1] | 213 | 9.7 | 104.5 | 9.2 | +| silero | gap5 | 229 [154, 321] | 0.4 [0.0, 1.1] | 149 | 9.7 | 101.8 | 9.2 | +| silero | gap2 | 185 [128, 255] | 0.5 [0.0, 1.3] | 168 | 9.7 | 101.8 | 9.2 | +| ultra | base | 317 [211, 445] | 0.0 [0.0, 0.1] | 147 | 32.0 | 89.6 | 31.5 | +| ultra | soft | 317 [211, 444] | 0.0 [0.0, 0.1] | 162 | 20.7 | 104.3 | 20.5 | +| ultra | longest | 258 [173, 360] | 0.2 [0.1, 0.3] | 195 | 32.0 | 90.3 | 31.5 | +| ultra | gap5 | 303 [204, 420] | 0.0 [0.0, 0.1] | 149 | 32.0 | 89.6 | 31.5 | +| ultra | gap2 | 270 [186, 369] | 0.1 [0.0, 0.1] | 164 | 32.0 | 89.6 | 31.5 | +| redux | base | 296 [192, 421] | 0.1 [0.0, 0.1] | 150 | 35.1 | 89.6 | 34.5 | +| redux | soft | 297 [193, 421] | 0.1 [0.0, 0.1] | 160 | 23.4 | 97.8 | 23.0 | +| redux | longest | 227 [148, 323] | 0.1 [0.1, 0.2] | 195 | 35.1 | 89.2 | 34.5 | +| redux | gap5 | 264 [174, 370] | 0.1 [0.0, 0.1] | 155 | 35.1 | 89.6 | 34.5 | +| redux | gap2 | 232 [157, 319] | 0.1 [0.0, 0.1] | 168 | 35.1 | 89.6 | 34.5 | +| fusion_redux | base | 253 [170, 356] | 0.1 [0.0, 0.3] | 150 | 39.2 | 86.5 | 38.1 | +| fusion_redux | soft | 257 [170, 363] | 0.1 [0.0, 0.3] | 150 | 38.7 | 86.5 | 37.6 | +| fusion_redux | longest | 191 [130, 267] | 0.3 [0.0, 0.7] | 188 | 39.2 | 85.4 | 38.1 | +| fusion_redux | gap5 | 226 [153, 316] | 0.3 [0.0, 0.7] | 154 | 39.2 | 86.5 | 38.1 | +| fusion_redux | gap2 | 198 [138, 274] | 0.4 [0.0, 1.0] | 167 | 39.2 | 86.5 | 38.1 | +| gate_redux | base | 238 [158, 336] | 0.9 [0.3, 1.7] | 152 | 34.2 | 93.7 | 32.9 | +| gate_redux | soft | 242 [161, 340] | 0.9 [0.3, 1.6] | 152 | 32.9 | 94.4 | 32.2 | +| gate_redux | longest | 172 [117, 245] | 1.6 [0.6, 2.9] | 193 | 34.2 | 95.5 | 32.9 | +| gate_redux | gap5 | 209 [141, 296] | 1.4 [0.5, 2.5] | 159 | 34.2 | 93.7 | 32.9 | +| gate_redux | gap2 | 174 [120, 245] | 1.7 [0.6, 3.0] | 177 | 34.2 | 93.7 | 32.9 | +| oracle | base | 230 [151, 324] | 0.0 [0.0, 0.0] | 149 | 55.2 | 53.8 | 53.8 | + +### ami (4.20 h) + +| detector | policy | non-speech s/h | speech lost % | segments/h | hard cuts/h | cuts in ref speech/h (all) | hard cuts in ref speech/h | +|---|---|---|---|---|---|---|---| +| silero | base | 355 [303, 427] | 5.3 [1.6, 11.6] | 133 | 4.8 | 100.5 | 1.2 | +| silero | soft | 351 [300, 422] | 5.5 [1.8, 11.7] | 134 | 0.0 | 102.1 | 0.0 | +| silero | longest | 176 [147, 204] | 9.0 [3.6, 17.8] | 206 | 4.8 | 105.7 | 1.2 | +| silero | gap5 | 252 [211, 299] | 7.4 [2.5, 15.6] | 150 | 4.8 | 100.5 | 1.2 | +| silero | gap2 | 175 [136, 212] | 9.0 [3.2, 18.6] | 205 | 4.8 | 100.5 | 1.2 | +| ultra | base | 598 [446, 806] | 0.6 [0.4, 1.0] | 137 | 0.5 | 90.5 | 0.2 | +| ultra | soft | 598 [446, 806] | 0.6 [0.4, 1.0] | 137 | 0.0 | 90.9 | 0.0 | +| ultra | longest | 349 [276, 445] | 2.3 [1.5, 3.5] | 218 | 0.5 | 90.0 | 0.2 | +| ultra | gap5 | 447 [367, 552] | 0.9 [0.6, 1.4] | 157 | 0.5 | 90.5 | 0.2 | +| ultra | gap2 | 318 [269, 384] | 2.0 [1.1, 3.5] | 225 | 0.5 | 90.5 | 0.2 | +| redux | base | 620 [453, 852] | 0.3 [0.2, 0.5] | 140 | 0.7 | 88.1 | 0.7 | +| redux | soft | 620 [452, 852] | 0.3 [0.2, 0.5] | 140 | 0.0 | 87.8 | 0.0 | +| redux | longest | 420 [320, 554] | 1.2 [0.7, 1.8] | 216 | 0.7 | 82.8 | 0.7 | +| redux | gap5 | 529 [402, 700] | 0.4 [0.3, 0.7] | 153 | 0.7 | 88.1 | 0.7 | +| redux | gap2 | 398 [318, 504] | 0.8 [0.4, 1.5] | 214 | 0.7 | 88.1 | 0.7 | +| fusion_redux | base | 360 [311, 425] | 4.1 [1.3, 8.6] | 141 | 3.8 | 93.3 | 1.7 | +| fusion_redux | soft | 357 [308, 422] | 4.1 [1.3, 8.8] | 142 | 1.2 | 94.0 | 1.2 | +| fusion_redux | longest | 209 [184, 237] | 6.5 [2.4, 13.4] | 207 | 3.8 | 94.8 | 1.7 | +| fusion_redux | gap5 | 275 [241, 313] | 5.6 [1.8, 12.0] | 155 | 3.8 | 93.3 | 1.7 | +| fusion_redux | gap2 | 219 [186, 252] | 6.4 [2.1, 13.8] | 191 | 3.8 | 93.3 | 1.7 | +| gate_redux | base | 426 [343, 551] | 2.1 [0.9, 4.1] | 144 | 1.7 | 94.5 | 0.7 | +| gate_redux | soft | 425 [341, 549] | 2.1 [0.9, 4.1] | 144 | 0.5 | 94.3 | 0.5 | +| gate_redux | longest | 229 [199, 272] | 5.3 [2.9, 9.3] | 211 | 1.7 | 105.5 | 0.7 | +| gate_redux | gap5 | 291 [257, 333] | 3.5 [1.4, 7.1] | 163 | 1.7 | 94.5 | 0.7 | +| gate_redux | gap2 | 225 [196, 256] | 5.3 [2.5, 9.8] | 214 | 1.7 | 94.5 | 0.7 | +| oracle | base | 375 [277, 523] | 0.0 [0.0, 0.0] | 155 | 34.3 | 32.6 | 32.6 | + +### ava (3.50 h) + +| detector | policy | non-speech s/h | speech lost % | segments/h | hard cuts/h | cuts in ref speech/h (all) | hard cuts in ref speech/h | +|---|---|---|---|---|---|---|---| +| silero | base | 885 [749, 1015] | 6.3 [3.7, 9.1] | 135 | 7.7 | 76.6 | 2.0 | +| silero | soft | 887 [752, 1018] | 6.3 [3.6, 9.1] | 135 | 0.0 | 78.6 | 0.0 | +| silero | longest | 403 [344, 460] | 10.9 [7.1, 15.0] | 219 | 7.7 | 56.6 | 2.0 | +| silero | gap5 | 521 [442, 606] | 9.9 [6.4, 13.6] | 178 | 7.7 | 76.6 | 2.0 | +| silero | gap2 | 361 [308, 420] | 11.1 [7.2, 15.2] | 255 | 7.7 | 76.6 | 2.0 | +| ultra | base | 1348 [1148, 1574] | 0.3 [0.1, 0.5] | 139 | 2.0 | 58.0 | 1.4 | +| ultra | soft | 1350 [1150, 1574] | 0.3 [0.1, 0.5] | 139 | 0.3 | 58.3 | 0.3 | +| ultra | longest | 1043 [881, 1236] | 1.4 [0.9, 2.0] | 225 | 2.0 | 41.1 | 1.4 | +| ultra | gap5 | 1206 [1030, 1410] | 0.7 [0.3, 1.2] | 158 | 2.0 | 58.0 | 1.4 | +| ultra | gap2 | 1036 [878, 1224] | 1.4 [0.7, 2.1] | 239 | 2.0 | 58.0 | 1.4 | +| redux | base | 1328 [1130, 1553] | 0.2 [0.1, 0.4] | 138 | 1.1 | 53.1 | 1.1 | +| redux | soft | 1328 [1129, 1552] | 0.2 [0.1, 0.4] | 138 | 0.9 | 53.4 | 0.9 | +| redux | longest | 998 [846, 1172] | 1.3 [0.9, 1.8] | 229 | 1.1 | 38.6 | 1.1 | +| redux | gap5 | 1176 [1009, 1366] | 0.6 [0.3, 0.9] | 159 | 1.1 | 53.1 | 1.1 | +| redux | gap2 | 974 [838, 1123] | 1.3 [0.7, 2.0] | 253 | 1.1 | 53.1 | 1.1 | +| fusion_redux | base | 869 [720, 1013] | 5.6 [3.0, 8.3] | 139 | 8.9 | 58.9 | 3.7 | +| fusion_redux | soft | 869 [719, 1013] | 5.6 [3.1, 8.3] | 139 | 0.6 | 60.0 | 0.6 | +| fusion_redux | longest | 477 [407, 545] | 9.1 [5.8, 12.5] | 212 | 8.9 | 49.4 | 3.7 | +| fusion_redux | gap5 | 583 [501, 670] | 8.3 [5.2, 11.5] | 175 | 8.9 | 58.9 | 3.7 | +| fusion_redux | gap2 | 445 [383, 511] | 9.2 [5.8, 12.7] | 242 | 8.9 | 58.9 | 3.7 | +| gate_redux | base | 832 [708, 954] | 7.8 [4.3, 11.6] | 147 | 7.1 | 63.7 | 3.1 | +| gate_redux | soft | 836 [717, 951] | 7.9 [4.4, 11.6] | 147 | 0.6 | 66.0 | 0.6 | +| gate_redux | longest | 430 [367, 492] | 11.5 [7.5, 15.8] | 221 | 7.1 | 61.7 | 3.1 | +| gate_redux | gap5 | 522 [442, 604] | 10.6 [6.8, 14.6] | 187 | 7.1 | 63.7 | 3.1 | +| gate_redux | gap2 | 390 [337, 446] | 12.1 [8.2, 16.4] | 256 | 7.1 | 63.7 | 3.1 | +| oracle | base | 833 [700, 962] | 0.0 [0.0, 0.0] | 152 | 17.4 | 11.7 | 11.7 | + +## S3. Where the decoded non-speech sits (trim 0.3, shipped policy), measured on the detector's own speech mask, seconds per hour: trim margins at segment ends, internal pauses of <1 s, 1-5 s and >= 5 s that stay inside a segment + +| domain | detector | edge margins | internal <1 s | internal 1-5 s | internal >= 5 s | total non-speech by detector mask | +|---|---|---|---|---|---|---| +| vox | silero | 44 | 223 | 119 | 56 | 441 | +| vox | ultra | 48 | 164 | 88 | 15 | 315 | +| vox | redux | 48 | 140 | 86 | 35 | 309 | +| vox | fusion_redux | 48 | 98 | 77 | 35 | 257 | +| vox | gate_redux | 53 | 95 | 92 | 49 | 290 | +| ami | silero | 59 | 327 | 336 | 173 | 894 | +| ami | ultra | 67 | 336 | 412 | 170 | 985 | +| ami | redux | 66 | 297 | 358 | 102 | 824 | +| ami | fusion_redux | 68 | 216 | 226 | 136 | 647 | +| ami | gate_redux | 72 | 190 | 292 | 187 | 741 | +| ava | silero | 64 | 264 | 449 | 467 | 1245 | +| ava | ultra | 65 | 353 | 466 | 163 | 1046 | +| ava | redux | 65 | 370 | 535 | 172 | 1142 | +| ava | fusion_redux | 70 | 204 | 365 | 366 | 1005 | +| ava | gate_redux | 75 | 158 | 360 | 396 | 990 | diff --git a/scripts/vad_bench/real_recordings/results/tuned.json b/scripts/vad_bench/real_recordings/results/tuned.json new file mode 100644 index 0000000..e70061f --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/tuned.json @@ -0,0 +1 @@ +{"sil": ["sil", 0.05], "silnat": ["nat", "silero", 0.1, 0.25, 0.5, 0.1], "('head', 'ultra')": ["head", "ultra", 0.35], "('or', 'ultra')": ["or", "ultra", 0.1, 0.9], "('and', 'ultra')": ["and", "ultra", 0.1, 0.1], "('mean', 'ultra')": ["mean", "ultra", 0.75, 0.25], "('two', 'ultra')": ["two", "ultra", 0.1, 0.3, 32, 8, 30], "('twog', 'ultra')": ["twog", "ultra", 0.92, 32, 16, 60], "('gate', 'ultra')": ["gate", "ultra", 0.85], "('head', 'redux')": ["head", "redux", 0.6], "('or', 'redux')": ["or", "redux", 0.1, 0.8], "('and', 'redux')": ["and", "redux", 0.1, 0.1], "('mean', 'redux')": ["mean", "redux", 0.75, 0.2], "('two', 'redux')": ["two", "redux", 0.1, 0.3, 32, 8, 30], "('twog', 'redux')": ["twog", "redux", 0.92, 32, 16, 60], "('gate', 'redux')": ["gate", "redux", 0.8]} \ No newline at end of file diff --git a/scripts/vad_bench/real_recordings/results/wer.md b/scripts/vad_bench/real_recordings/results/wer.md new file mode 100644 index 0000000..69fbde5 --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/wer.md @@ -0,0 +1,25 @@ +# WER (ted), decoding the segments of each system + +## ted: ultra ASR model (VAD segmentation from ultra head / Silero / fusion / gate) + +| system | WER % [95% CI] | S | D | I | decoded s | per-talk WER % | dWER vs head (pp) | dWER vs silero (pp) | +|---|---|---|---|---|---|---|---|---| +| head | 4.07 [3.40, 4.79] | 191 | 104 | 65 | 2882 | 3.1, 2.2, 5.1, 5.7, 3.2 | +0.00 [+0.00, +0.00] | -0.10 [-0.38, +0.16] | +| silero | 4.18 [3.48, 4.88] | 196 | 107 | 66 | 2887 | 3.4, 2.6, 5.0, 5.7, 3.2 | +0.10 [-0.16, +0.38] | +0.00 [+0.00, +0.00] | +| fusion_def | 4.18 [3.49, 4.88] | 196 | 112 | 61 | 2882 | 3.0, 2.2, 5.0, 5.7, 3.5 | +0.10 [-0.08, +0.33] | +0.00 [-0.21, +0.22] | +| fusion_tuned | 4.21 [3.52, 4.92] | 192 | 107 | 73 | 2889 | 3.1, 2.1, 4.9, 6.0, 3.5 | +0.14 [-0.11, +0.39] | +0.03 [-0.22, +0.29] | +| gate92 | 4.24 [3.55, 4.96] | 200 | 109 | 66 | 2881 | 3.1, 2.2, 5.0, 6.0, 3.5 | +0.17 [-0.03, +0.42] | +0.07 [-0.11, +0.27] | +| orgate92 | 4.19 [3.50, 4.89] | 196 | 112 | 62 | 2882 | 3.0, 2.2, 5.1, 5.7, 3.5 | +0.11 [-0.07, +0.33] | +0.01 [-0.20, +0.23] | + +## ted: redux ASR model (VAD segmentation from redux head / Silero / fusion / gate) + +| system | WER % [95% CI] | S | D | I | decoded s | per-talk WER % | dWER vs head (pp) | dWER vs silero (pp) | +|---|---|---|---|---|---|---|---|---| +| head | 5.02 [4.31, 5.75] | 253 | 119 | 72 | 2882 | 3.9, 4.7, 6.0, 6.5, 3.8 | +0.00 [+0.00, +0.00] | +0.00 [-0.26, +0.23] | +| silero | 5.02 [4.23, 5.81] | 250 | 119 | 75 | 2887 | 3.6, 4.5, 6.2, 6.9, 3.6 | +0.00 [-0.23, +0.26] | +0.00 [+0.00, +0.00] | +| fusion_def | 5.02 [4.31, 5.73] | 251 | 118 | 75 | 2883 | 4.1, 4.9, 5.8, 6.4, 3.9 | +0.00 [-0.16, +0.16] | +0.00 [-0.28, +0.28] | +| fusion_tuned | 4.97 [4.23, 5.72] | 256 | 116 | 67 | 2890 | 3.7, 4.3, 5.4, 6.6, 4.0 | -0.06 [-0.32, +0.19] | -0.06 [-0.35, +0.22] | +| gate92 | 4.97 [4.24, 5.69] | 246 | 119 | 74 | 2881 | 3.9, 4.9, 6.0, 6.2, 3.8 | -0.06 [-0.23, +0.11] | -0.06 [-0.34, +0.20] | +| orgate92 | 5.02 [4.32, 5.71] | 248 | 120 | 76 | 2883 | 4.1, 4.9, 5.9, 6.4, 3.9 | +0.00 [-0.16, +0.16] | +0.00 [-0.27, +0.24] | + +talk order: GaryFlake-merged, RobertGupta-merged, EricMead_2009P_EricMead-merged, DanBarber-merged, MichaelSpecter-merged diff --git a/scripts/vad_bench/real_recordings/results/wer_comp.md b/scripts/vad_bench/real_recordings/results/wer_comp.md new file mode 100644 index 0000000..92359f3 --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/wer_comp.md @@ -0,0 +1,25 @@ +# WER (comp), decoding the segments of each system + +## comp: ultra ASR model (VAD segmentation from ultra head / Silero / fusion / gate) + +| system | WER % [95% CI] | S | D | I | decoded s | per-talk WER % | dWER vs head (pp) | dWER vs silero (pp) | +|---|---|---|---|---|---|---|---|---| +| head | 5.91 [3.66, 8.75] | 64 | 29 | 113 | 1377 | 5.4, 4.9, 6.8 | +0.00 [+0.00, +0.00] | +2.04 [-0.09, +4.54] | +| silero | 3.87 [2.91, 4.90] | 63 | 30 | 42 | 1137 | 3.4, 2.1, 5.2 | -2.04 [-4.54, +0.09] | +0.00 [+0.00, +0.00] | +| fusion_def | 4.16 [3.18, 5.22] | 66 | 36 | 43 | 1146 | 3.7, 3.0, 5.2 | -1.75 [-4.02, +0.09] | +0.29 [-0.11, +0.73] | +| fusion_tuned | 5.91 [3.71, 8.75] | 65 | 29 | 112 | 1215 | 6.0, 5.1, 6.3 | +0.00 [-0.43, +0.40] | +2.04 [-0.05, +4.53] | +| gate92 | 4.24 [2.97, 5.79] | 63 | 32 | 53 | 1160 | 4.9, 2.1, 5.0 | -1.66 [-3.98, +0.00] | +0.37 [-0.32, +1.54] | +| orgate92 | 4.36 [3.14, 5.84] | 64 | 32 | 56 | 1169 | 4.8, 2.5, 5.1 | -1.55 [-3.81, +0.11] | +0.49 [-0.20, +1.62] | + +## comp: redux ASR model (VAD segmentation from redux head / Silero / fusion / gate) + +| system | WER % [95% CI] | S | D | I | decoded s | per-talk WER % | dWER vs head (pp) | dWER vs silero (pp) | +|---|---|---|---|---|---|---|---|---| +| head | 6.94 [4.76, 9.55] | 91 | 38 | 113 | 1367 | 5.6, 7.3, 7.7 | +0.00 [+0.00, +0.00] | +2.18 [+0.20, +4.66] | +| silero | 4.76 [3.85, 5.71] | 88 | 39 | 39 | 1137 | 3.4, 4.7, 5.8 | -2.18 [-4.66, -0.20] | +0.00 [+0.00, +0.00] | +| fusion_def | 5.02 [4.03, 6.05] | 96 | 33 | 46 | 1147 | 4.2, 4.9, 5.7 | -1.92 [-4.24, -0.08] | +0.26 [-0.11, +0.62] | +| fusion_tuned | 6.25 [4.47, 8.66] | 94 | 35 | 89 | 1224 | 5.4, 7.6, 6.1 | -0.69 [-2.03, +0.18] | +1.49 [+0.00, +3.54] | +| gate92 | 5.22 [4.25, 6.24] | 97 | 37 | 48 | 1143 | 4.2, 5.0, 6.1 | -1.72 [-3.94, +0.08] | +0.46 [+0.03, +0.93] | +| orgate92 | 5.62 [4.29, 7.15] | 93 | 36 | 67 | 1167 | 4.0, 6.2, 6.5 | -1.32 [-2.95, -0.06] | +0.86 [-0.03, +2.05] | + +talk order: GaryFlake+ins, RobertGupta+ins, EricMead_2009P_EricMead+ins diff --git a/scripts/vad_bench/real_recordings/results/wer_forced.md b/scripts/vad_bench/real_recordings/results/wer_forced.md new file mode 100644 index 0000000..201c661 --- /dev/null +++ b/scripts/vad_bench/real_recordings/results/wer_forced.md @@ -0,0 +1,17 @@ +# WER (forced), decoding the segments of each system + +## forced: ultra ASR model (VAD segmentation from ultra head / Silero / fusion / gate) + +| system | WER % [95% CI] | S | D | I | decoded s | per-talk WER % | dWER vs pause20 (pp) | dWER vs forced20 (pp) | +|---|---|---|---|---|---|---|---|---| +| pause20 | 4.19 [3.46, 4.94] | 195 | 103 | 72 | 2885 | 3.3, 2.1, 5.4, 5.5, 3.4 | +0.00 [+0.00, +0.00] | -0.29 [-0.61, +0.00] | +| forced20 | 4.48 [3.76, 5.24] | 202 | 125 | 69 | 2894 | 3.7, 3.2, 4.6, 6.3, 3.6 | +0.29 [-0.00, +0.61] | +0.00 [+0.00, +0.00] | + +## forced: redux ASR model (VAD segmentation from redux head / Silero / fusion / gate) + +| system | WER % [95% CI] | S | D | I | decoded s | per-talk WER % | dWER vs pause20 (pp) | dWER vs forced20 (pp) | +|---|---|---|---|---|---|---|---|---| +| pause20 | 5.15 [4.39, 5.92] | 255 | 123 | 77 | 2885 | 3.7, 4.8, 5.8, 6.8, 4.2 | +0.00 [+0.00, +0.00] | -0.32 [-0.65, +0.05] | +| forced20 | 5.47 [4.66, 6.28] | 264 | 145 | 74 | 2894 | 3.9, 5.4, 5.8, 7.6, 4.2 | +0.32 [-0.05, +0.65] | +0.00 [+0.00, +0.00] | + +talk order: GaryFlake-merged, RobertGupta-merged, EricMead_2009P_EricMead-merged, DanBarber-merged, MichaelSpecter-merged diff --git a/scripts/vad_bench/real_recordings/rl.py b/scripts/vad_bench/real_recordings/rl.py new file mode 100644 index 0000000..e9e012e --- /dev/null +++ b/scripts/vad_bench/real_recordings/rl.py @@ -0,0 +1,44 @@ +"""results loader + bootstrap helpers""" +import pickle, numpy as np, os, sys +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import vr +R = pickle.load(open(f"{vr.ROOT}/results/counts.pkl", "rb")) +SP = R["specs"]; C = R["counts"]; M = R["meta"]; IDX = {s: i for i, s in enumerate(SP)} +SPEECH_DOM = ("vox", "ami", "ava"); NS_DOM = ("music", "noise", "esc") +def sel(domains=None, split=None): + return np.array([i for i, m in enumerate(M) if (domains is None or m["domain"] in domains) and (split is None or m["split"] == split)]) +def f1_of(c): + tp, fp, fn = c[..., 0], c[..., 1], c[..., 2] + P = tp / np.maximum(1, tp + fp); Rr = tp / np.maximum(1, tp + fn) + return P, Rr, 2 * P * Rr / np.maximum(1e-12, P + Rr) +def macro_f1(spec, split): + f = [] + for d in SPEECH_DOM: + ii = sel((d,), split); c = C[ii, IDX[spec], :4].sum(0); f.append(f1_of(c)[2]) + return float(np.mean(f)) +def tune(cands, split="tune"): + best = max(cands, key=lambda s: macro_f1(s, split)); return best +B = 2000 +_rng = np.random.default_rng(12345) +_BI = {} +def boot_indices(ii_by_group): + """stratified bootstrap indices, cached per group-key so that every system uses the same resamples (paired).""" + key = tuple(tuple(g) for g in ii_by_group) + if key not in _BI: + _BI[key] = [ii[_rng.integers(0, len(ii), size=(B, len(ii)))] for ii in ii_by_group] + return _BI[key] +def boot_counts(spec, groups): + """returns (B,4) bootstrap pooled counts (sum over groups) and point counts (4,)""" + bi = boot_indices(groups); k = IDX[spec] + tot = sum(C[b, k, :4].sum(1) for b in bi) + pt = sum(C[g, k, :4].sum(0) for g in groups) + return tot, pt +def ci(x, lo=2.5, hi=97.5): return np.percentile(x, [lo, hi]) +def fmt(point, bs, scale=100, d=1): + l, h = ci(bs); return f"{point*scale:.{d}f} [{l*scale:.{d}f}, {h*scale:.{d}f}]" +def stat(spec, groups, which=2): + tot, pt = boot_counts(spec, groups) + return f1_of(pt)[which], f1_of(tot)[which] +def delta(spec_a, spec_b, groups, which=2): + ta, pa = boot_counts(spec_a, groups); tb, pb = boot_counts(spec_b, groups) + return f1_of(pa)[which] - f1_of(pb)[which], f1_of(ta)[which] - f1_of(tb)[which] diff --git a/scripts/vad_bench/real_recordings/seg_eval.py b/scripts/vad_bench/real_recordings/seg_eval.py new file mode 100644 index 0000000..3bc3d3b --- /dev/null +++ b/scripts/vad_bench/real_recordings/seg_eval.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +"""Seconds of non-speech sent to the decoder by the transcribe --vad segmentation (segment_by_vad replica, trim .3 default), per detector/system and cut policy. +-> results/seg.pkl : per recording, per system: [decoded_s, decoded_nonspeech_s, speech_s, speech_in_segments_s, n_segments, hard_cuts]""" +import os, sys, pickle, time +import numpy as np +from multiprocessing import Pool +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import vr, systems +from vr import FS +FUS = {"ultra": ("two", "ultra", 0.5, 0.5, 16, 16, 30), "redux": ("two", "redux", 0.5, 0.5, 16, 16, 30)} +def system_masks(r): + """name -> (smoothed speech mask, period, min_pause)""" + out = {} + out["silero"] = (vr.smooth_native(r.P["silero"], FS["silero"], r.dur, 0.5, 0.1, 0.25), FS["silero"], 0.1) + for t in (0.1, 0.2, 0.3): + out[f"silero_t{int(t*10):02d}"] = (vr.smooth_native(r.P["silero"], FS["silero"], r.dur, t, 0.1, 0.25), FS["silero"], 0.1) + out["silero_p02"] = (vr.smooth_native(r.P["silero"], FS["silero"], r.dur, 0.5, 0.1, 0.25), FS["silero"], 0.2) + for h in ("ultra", "redux"): + out[h] = (vr.smooth_native(r.P[h], FS[h], r.dur, 0.5, 0.1, 0.1), FS[h], 0.2) + out[f"{h}_t09"] = (vr.smooth_native(r.P[h], FS[h], r.dur, 0.9, 0.1, 0.1), FS[h], 0.2) + out[f"fusion_{h}"] = (systems.mask(r, FUS[h]), vr.GR, 0.2) + out[f"fusiontuned_{h}"] = (systems.mask(r, ("two", h, 0.1, 0.3, 32, 8, 30)), vr.GR, 0.2) + out[f"gate98_{h}"] = (systems.mask(r, ("gate", h, 0.98)), vr.GR, 0.2) + out[f"gate_{h}"] = (systems.mask(r, ("gate", h, 0.92)), vr.GR, 0.2) + out[f"orgate_{h}"] = (systems.mask(r, ("org", h, 0.92)), vr.GR, 0.2) + out["oracle"] = (systems.mask(r, ("ref",)), vr.GR, 0.2) + return out +POL = [("base", dict()), ("longest", dict(policy="longest")), ("soft", dict(policy="soft")), ("gap2", dict(long_gap=2.0)), ("gap5", dict(long_gap=5.0))] +def work(meta): + r = systems.Rec(meta); ref = meta["ref"]; n = r.n; res = {} + M = system_masks(r) + for name, (sp, fs, mp) in M.items(): + for pol, kw in POL: + if pol != "base" and name not in ("silero", "ultra", "redux", "fusion_ultra", "fusion_redux", "gate_ultra", "gate_redux"): continue + for trim in (0.3, 0.0): + if trim == 0.0 and pol != "base": continue + segs, hard = vr.segment_by_mask(sp, fs, r.dur, min_pause=mp, trim=trim, **kw) + sm = vr.seg_mask(segs, n) + cuts = list(vr.LAST_CUTS) + in_sp = sum(1 for c, hd in cuts if ref[min(n - 1, int(c / vr.GR))]); hard_in_sp = sum(1 for c, hd in cuts if hd and ref[min(n - 1, int(c / vr.GR))]) + # decoded seconds of detector-non-speech by gap class (edge = trim margin at a segment end; internal gaps by length) + dec = np.zeros(5) # edge, int<1s, 1-5s, >=5s, total + if trim == 0.3 and pol == "base": + spm = vr.hold(sp.astype(np.float32), fs, n) >= 0.5 if fs != vr.GR else sp[:n] + for a, b in segs: + ia, ib = int(round(a / vr.GR)), int(round(b / vr.GR)); seg = spm[ia:ib] + if len(seg) == 0: continue + ss, ee = vr.runs(~seg) + for x, y in zip(ss, ee): + L = (y - x) * vr.GR + if x == 0 or y == len(seg): dec[0] += L + elif L < 1: dec[1] += L + elif L < 5: dec[2] += L + else: dec[3] += L + dec[4] += (~seg).sum() * vr.GR + res[(name, pol, trim)] = np.concatenate([[sm.sum() * vr.GR, (sm & ~ref).sum() * vr.GR, ref.sum() * vr.GR, (sm & ref).sum() * vr.GR, len(segs), hard, len(cuts), in_sp, hard_in_sp], dec]) + return meta["id"], res +if __name__ == "__main__": + D = vr.load_recordings(); t = time.time() + with Pool(8) as p: res = dict(p.map(work, D, chunksize=1)) + pickle.dump(dict(meta=[dict(id=m["id"], domain=m["domain"], split=m["split"], dur=m["dur"], nonspeech=m["nonspeech"]) for m in D], res=res), open(f"{vr.ROOT}/results/seg.pkl", "wb")) + print("done", time.time() - t) diff --git a/scripts/vad_bench/real_recordings/seg_report.py b/scripts/vad_bench/real_recordings/seg_report.py new file mode 100644 index 0000000..2bf7a23 --- /dev/null +++ b/scripts/vad_bench/real_recordings/seg_report.py @@ -0,0 +1,49 @@ +#!/usr/bin/env python3 +import pickle, numpy as np, sys, os +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import vr +R = pickle.load(open(f"{vr.ROOT}/results/seg.pkl", "rb")); M = R["meta"]; res = R["res"] +DOMS = ["vox", "ami", "ava", "music", "noise", "esc"] +rng = np.random.default_rng(11) +IDS = {d: [m["id"] for m in M if m["domain"] == d] for d in DOMS}; HRS = {d: sum(m["dur"] for m in M if m["domain"] == d) / 3600 for d in DOMS} +BI = {d: rng.integers(0, len(IDS[d]), size=(2000, len(IDS[d]))) for d in DOMS} +def vals(dom, key, col): + return np.array([res[i][key][col] for i in IDS[dom]]), np.array([m["dur"] for m in M if m["domain"] == dom]) / 3600 +def rate(dom, key, col, per="hour"): + v, h = vals(dom, key, col); est = v.sum() / h.sum(); bs = v[BI[dom]].sum(1) / h[BI[dom]].sum(1) + l, u = np.percentile(bs, [2.5, 97.5]); return est, l, u +def lost(dom, key): + sp, _ = vals(dom, key, 2); ins, _ = vals(dom, key, 3) + if sp.sum() == 0: return None + est = 1 - ins.sum() / sp.sum(); bs = 1 - ins[BI[dom]].sum(1) / sp[BI[dom]].sum(1); l, u = np.percentile(bs, [2.5, 97.5]); return est, l, u +f = lambda t, d=0: f"{t[0]:.{d}f} [{t[1]:.{d}f}, {t[2]:.{d}f}]" +out = ["# Segmentation: non-speech sent to the decoder (segment_by_vad replica, verified against the CLI), per hour of audio", ""] +SYSN = ["silero", "silero_t03", "silero_t02", "silero_t01", "ultra", "redux", "ultra_t09", "redux_t09", "fusion_ultra", "fusion_redux", "fusiontuned_ultra", "fusiontuned_redux", "gate_ultra", "gate_redux", "gate98_redux", "orgate_ultra", "orgate_redux", "oracle"] +out += ["## S1. Seconds of reference non-speech inside the decoded segments, per hour of audio (95% CI over recordings), old behaviour (trim 0) -> default (trim 0.3). Last column pair: share of reference speech outside the segments (speech lost), trim 0.3", ""] +out += ["| system | " + " | ".join(f"{d} trim0 | {d} trim.3 | {d} lost %" for d in DOMS[:3]) + " | " + " | ".join(f"{d} trim0 | {d} trim.3" for d in DOMS[3:]) + " |", "|---|" + "---|" * (9 + 6)] +for s in SYSN: + row = [] + for d in DOMS[:3]: + a = rate(d, (s, "base", 0.0), 1); b = rate(d, (s, "base", 0.3), 1); lo = lost(d, (s, "base", 0.3)) + row += [f(a), f(b), f"{100*lo[0]:.1f} [{100*lo[1]:.1f}, {100*lo[2]:.1f}]"] + for d in DOMS[3:]: + a = rate(d, (s, "base", 0.0), 1); b = rate(d, (s, "base", 0.3), 1); row += [f(a), f(b)] + out.append(f"| {s} | " + " | ".join(row) + " |") +out += ["", "Reading: for music/noise/esc every decoded second is non-speech; 3600 means the whole hour was decoded. Short files (<= 30 s) are never cut or trimmed by the segmenter (not present here: all files are >= 6 min).", ""] +out += ["## S2. Cut policy (trim 0.3): non-speech seconds decoded per hour [95% CI], speech lost %, number of segments per hour, hard cuts per hour and cuts that land in reference speech per hour", ""] +POLS = ["base", "soft", "longest", "gap5", "gap2"] +out += ["policies: base = shipped rule; soft = shipped rule, but before a hard cut take the midpoint of the longest silent gap of any length in the window; longest = cut at the longest pause (>= min pause) in the window; gap5 / gap2 = shipped rule plus a cut at every pause of at least 5 s / 2 s (decoded segments can then be shorter than 1 s apart).", ""] +for d in DOMS[:3]: + out += [f"### {d} ({HRS[d]:.2f} h)", "", "| detector | policy | non-speech s/h | speech lost % | segments/h | hard cuts/h | cuts in ref speech/h (all) | hard cuts in ref speech/h |", "|---|---|---|---|---|---|---|---|"] + for s in ("silero", "ultra", "redux", "fusion_redux", "gate_redux", "oracle"): + for p in POLS: + k = (s, p, 0.3) + if k not in res[IDS[d][0]]: continue + lo = lost(d, k) + out.append(f"| {s} | {p} | {f(rate(d, k, 1))} | {100*lo[0]:.1f} [{100*lo[1]:.1f}, {100*lo[2]:.1f}] | {rate(d, k, 4)[0]:.0f} | {rate(d, k, 5)[0]:.1f} | {rate(d, k, 7)[0]:.1f} | {rate(d, k, 8)[0]:.1f} |") + out.append("") +out += ["## S3. Where the decoded non-speech sits (trim 0.3, shipped policy), measured on the detector's own speech mask, seconds per hour: trim margins at segment ends, internal pauses of <1 s, 1-5 s and >= 5 s that stay inside a segment", "", "| domain | detector | edge margins | internal <1 s | internal 1-5 s | internal >= 5 s | total non-speech by detector mask |", "|---|---|---|---|---|---|---|"] +for d in DOMS[:3]: + for s in ("silero", "ultra", "redux", "fusion_redux", "gate_redux"): + k = (s, "base", 0.3); out.append(f"| {d} | {s} | " + " | ".join(f"{rate(d, k, c)[0]:.0f}" for c in (9, 10, 11, 12, 13)) + " |") +open(f"{vr.ROOT}/results/seg.md", "w").write("\n".join(out) + "\n"); print("\n".join(out)) diff --git a/scripts/vad_bench/real_recordings/systems.py b/scripts/vad_bench/real_recordings/systems.py new file mode 100644 index 0000000..b1db5e4 --- /dev/null +++ b/scripts/vad_bench/real_recordings/systems.py @@ -0,0 +1,55 @@ +"""System definitions: spec tuple -> final 10 ms speech mask for one recording. + +A recording object `r` has: n (10 ms cells), dur, P = {silero,ultra,redux} native-frame probabilities, G = hold()-grid of each. +All 'unified' systems use the same post (bridge .1, min speech .1, min pause .2, pad 0); 'nat' systems use each detector's own options +through the exact native-frame replica of speech_regions().""" +import numpy as np +import vr +from vr import FS, GR + +class Rec: + def __init__(self, meta): + self.m = meta; self.id = meta["id"]; self.n = len(meta["ref"]); self.dur = meta["dur"] + self.P = vr.load_probs(self.id) + self.G = {k: vr.hold(self.P[k], FS[k], self.n) for k in FS} + self.gate_cache = {} + def gated(self, h, med): + """gated head probability on the grid: p where the raw p>=0.5 run has median >= med, else 0""" + key = (h, med) + if key not in self.gate_cache: + keep = vr.gate_frames(self.P[h], 0.5, med) + g = np.where(keep, self.P[h], 0).astype(np.float32) + self.gate_cache[key] = vr.hold(g, FS[h], self.n) + return self.gate_cache[key] + +def mask(r, spec, pp=None): + pp = vr.POST_HEAD if pp is None else pp + k = spec[0] + if k == "pp": # ("pp", min_speech, min_pause, pad, inner_spec): inner system with its own post-processing + return mask(r, spec[4], dict(bridge=0.1, min_speech=spec[1], min_pause=spec[2], pad=spec[3])) + if k == "sil": # unified post: ("sil", thr) + return vr.post(r.G["silero"] >= spec[1], **pp) + if k == "head": # ("head", model, thr) + return vr.post(r.G[spec[1]] >= spec[2], **pp) + if k == "nat": # native replica: ("nat", model, thr, min_speech, min_pause, pad) + _, h, thr, ms, mp, pad = spec + return vr.native_mask(r.P[h], FS[h], r.dur, thr, 0.1, ms, mp, pad, r.n)[0] + if k == "or": # ("or", model, ts, th) + return vr.post((r.G["silero"] >= spec[2]) | (r.G[spec[1]] >= spec[3]), **pp) + if k == "and": + return vr.post((r.G["silero"] >= spec[2]) & (r.G[spec[1]] >= spec[3]), **pp) + if k == "mean": # ("mean", model, w, thr) + return vr.post((spec[2] * r.G["silero"] + (1 - spec[2]) * r.G[spec[1]]) >= spec[3], **pp) + if k == "two": # ("two", model, ts, th, pre, post, Y) cells of 10 ms + _, h, ts, th, pre, po, Y = spec + return vr.post(vr.two_stage(r.G["silero"] >= ts, r.G[h] >= th, pre, po, Y), **pp) + if k == "gate": # ("gate", model, med) head thr .5 with run median gate + return vr.post(r.gated(spec[1], spec[2]) >= 0.5, **pp) + if k == "twog": # two-stage whose head input is the gated head: ("twog", model, med, pre, post, Y) + _, h, med, pre, po, Y = spec + return vr.post(vr.two_stage(r.G["silero"] >= 0.5, r.gated(h, med) >= 0.5, pre, po, Y), **pp) + if k == "org": # OR(Silero .5, gated head): ("org", model, med) + return vr.post((r.G["silero"] >= 0.5) | (r.gated(spec[1], spec[2]) >= 0.5), **pp) + if k == "ref": # oracle: the reference mask itself (post applied to keep the same smoothing) + return vr.post(r.m["ref"], **pp) + raise ValueError(spec) diff --git a/scripts/vad_bench/real_recordings/verify_gate.py b/scripts/vad_bench/real_recordings/verify_gate.py new file mode 100644 index 0000000..8f5c596 --- /dev/null +++ b/scripts/vad_bench/real_recordings/verify_gate.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +"""Check the C++ run gate (SegmenterOpts::run_gate) against the study's Python gate on the same probabilities. + +For every recording and detector in ROOT/probs it applies the gate in Python (vr.gate_frames: median of the +frames of each raw run of p >= 0.5, kept when >= gate), forms the speech regions with the repository's rules +(vr.native_mask), and compares them with the output of `gate_segtool` (built from gate_segtool.cpp and +src/vad_segmenter.cpp) on the same probabilities. + +Build the tool from the repository root: + g++ -O2 -std=c++17 -Isrc scripts/vad_bench/real_recordings/gate_segtool.cpp src/vad_segmenter.cpp -o gate_segtool +Run (VAD_REAL_ROOT holds probs/, as made by dump_probs.py; GATE_SEGTOOL is the binary; GATE_TMP a scratch dir): + VAD_REAL_ROOT=work GATE_SEGTOOL=./gate_segtool python3 scripts/vad_bench/real_recordings/verify_gate.py +Prints the number of comparisons, the mismatches and the seconds of speech the gate removes per domain.""" +import collections, glob, json, os, subprocess, sys, tempfile +import numpy as np +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import vr + +TOOL = os.environ["GATE_SEGTOOL"] +OPT = {"silero": (0.25, 0.1, 0.03), "ultra": (0.1, 0.2, 0.0), "redux": (0.1, 0.2, 0.0)} # min speech, min pause, pad +GATES = {"ultra": [0.92, 0.96, 0.5], "redux": [0.92, 0.96, 0.98], "silero": [0.9, 0.97]} +tmp = tempfile.mkdtemp(prefix="gate_", dir=os.environ.get("GATE_TMP")) +raw = os.path.join(tmp, "p.f32") + +def cpp(p, dur, kind, gate): + p.astype(np.float32).tofile(raw) + out = subprocess.run([TOOL, raw, repr(dur), "silero" if kind == "silero" else "head", repr(gate), "speech"], + capture_output=True, text=True, check=True).stdout.split() + return np.array(out, float).reshape(-1, 2) + +ids = sorted({os.path.basename(f).rsplit(".", 2)[0] for f in glob.glob(f"{vr.ROOT}/probs/*.npy")}) +total = bad = 0 +dropped = collections.defaultdict(lambda: [0.0, 0.0, 0]) # domain -> [C++ s, Python s, recordings] for Redux at 0.92 +for rid in ids: + for kind in ("ultra", "redux", "silero"): + pj = [f"{vr.ROOT}/probs/{rid}.{k}.json" for k in (kind, "silero", "redux", "ultra")] + pj = next((f for f in pj if os.path.exists(f)), None) + if pj is None: continue + dur = json.load(open(pj))["duration"] + p = np.load(f"{vr.ROOT}/probs/{rid}.{kind}.npy").astype(np.float32) + ms, mp, pad = OPT[kind] + n10 = int(round(dur / vr.GR)) + for g in GATES[kind]: + keep = vr.gate_frames(p, 0.5, g) + _, st, en = vr.native_mask(np.where(keep, p, 0).astype(np.float32), vr.FS[kind], dur, 0.5, 0.1, ms, mp, pad, n10) + c = cpp(p, dur, kind, g) + ok = len(c) == len(st) and (len(c) == 0 or (np.abs(c[:, 0] - np.array(st)).max() < 1e-6 and np.abs(c[:, 1] - np.array(en)).max() < 1e-6)) + total += 1 + if not ok: + bad += 1; print("MISMATCH", rid, kind, g, len(c), len(st)) + if kind == "redux": + c0, c1 = cpp(p, dur, kind, 0.0), cpp(p, dur, kind, 0.92) + _, s0, e0 = vr.native_mask(p, 0.08, dur, 0.5, 0.1, 0.1, 0.2, 0.0, n10) + keep = vr.gate_frames(p, 0.5, 0.92) + _, s1, e1 = vr.native_mask(np.where(keep, p, 0).astype(np.float32), 0.08, dur, 0.5, 0.1, 0.1, 0.2, 0.0, n10) + a = dropped[rid.split("_")[0]] + a[0] += (c0[:, 1] - c0[:, 0]).sum() - (c1[:, 1] - c1[:, 0]).sum() + a[1] += (np.array(e0) - np.array(s0)).sum() - (np.array(e1) - np.array(s1)).sum() + a[2] += 1 +print("compared", total, "mismatches", bad) +for d, a in sorted(dropped.items()): + print(f"{d}: {a[2]} recordings, Redux head at 0.92 removes {a[0]:.3f} s of speech (C++) and {a[1]:.3f} s (Python)") diff --git a/scripts/vad_bench/real_recordings/verify_native.py b/scripts/vad_bench/real_recordings/verify_native.py new file mode 100644 index 0000000..cd6fe35 --- /dev/null +++ b/scripts/vad_bench/real_recordings/verify_native.py @@ -0,0 +1,13 @@ +import json,numpy as np,vr,sys +D=vr.load_recordings(); bad=0;tot=0;maxd=0 +OPT={"silero":(0.25,0.1,0.03),"ultra":(0.1,0.2,0.0),"redux":(0.1,0.2,0.0)} +for m in D: + P=vr.load_probs(m["id"]) + for k in vr.FS: + j=json.load(open(f"{vr.ROOT}/probs/{m['id']}.{k}.json")) + ms,mp,pad=OPT[k] + _,st,en=vr.native_mask(P[k],vr.FS[k],j["duration"],0.5,0.1,ms,mp,pad,int(round(m["dur"]/vr.GR))) + cli=j["segments"]; tot+=1 + ok=len(cli)==len(st) and all(abs(c["start"]-a)<=0.0006 and abs(c["end"]-b)<=0.0006 for c,a,b in zip(cli,st,en)) + if not ok: bad+=1; print("MISMATCH",m["id"],k,len(cli),len(st)) +print("checked",tot,"mismatch",bad) diff --git a/scripts/vad_bench/real_recordings/verify_seg.py b/scripts/vad_bench/real_recordings/verify_seg.py new file mode 100644 index 0000000..0c91bb5 --- /dev/null +++ b/scripts/vad_bench/real_recordings/verify_seg.py @@ -0,0 +1,20 @@ +"""Compare the Python segmenter replica with `parakeet-cli vad --mode segments` (clean master build) on every recording and detector.""" +import json, subprocess, sys, numpy as np +import vr +D = vr.load_recordings(); MOD = {"silero": "silero-vad-f16", "ultra": "ultra-vad-q8_0", "redux": "redux-vad"} +OPT = {"silero": (0.25, 0.1), "ultra": (0.1, 0.2), "redux": (0.1, 0.2)} +files = [m for i, m in enumerate(D) if i % int(sys.argv[1] if len(sys.argv) > 1 else 6) == 0] +tot = bad = 0 +for m in files: + P = vr.load_probs(m["id"]) + for k in MOD: + for trim in (0.3, 0.0): + r = subprocess.run([vr.CLI_CLEAN, "vad", "--model", f"{vr.MODELS}/{MOD[k]}.gguf", "--input", m["wav"], "--mode", "segments", "--trim", str(trim), "--threads", "4"], capture_output=True, text=True) + j = json.loads(r.stdout); cli = [(s["start"], s["end"]) for s in j["segments"]] + ms, mp = OPT[k] + sp = vr.smooth_native(P[k], vr.FS[k], m["dur"], 0.5, 0.1, ms) + mine, hard = vr.segment_by_mask(sp, vr.FS[k], m["dur"], min_pause=mp, trim=trim) + ok = len(cli) == len(mine) and all(abs(a - c) < 1.1e-3 and abs(b - d) < 1.1e-3 for (a, b), (c, d) in zip(cli, mine)) + tot += 1; bad += not ok + if not ok: print("MISMATCH", m["id"], k, trim, len(cli), len(mine)) +print("checked", tot, "mismatch", bad) diff --git a/scripts/vad_bench/real_recordings/vr.py b/scripts/vad_bench/real_recordings/vr.py new file mode 100644 index 0000000..ca50324 --- /dev/null +++ b/scripts/vad_bench/real_recordings/vr.py @@ -0,0 +1,371 @@ +"""Shared library for the real-recording VAD study. 10 ms evaluation grid; numba kernels. + +Segmenter semantics are those of parakeet.cpp src/vad_segmenter.cpp (master da8bb4c).""" +import glob, json, os, hashlib, math +import numpy as np +from numba import njit + +ROOT = os.environ.get("VAD_REAL_ROOT") or os.getcwd() +GR = 0.01 +CLI_CLEAN = os.environ.get("PARAKEET_CLI", f"{ROOT}/parakeet-cli-clean") # a parakeet-cli of master (no run gate needed) +MODELS = os.environ.get("VAD_REAL_MODELS", f"{ROOT}/models") # silero-vad-f16, ultra-vad-q8_0, redux-vad GGUFs (and ASR GGUFs for wer_dec.py) +FS = {"silero": 0.032, "ultra": 0.08, "redux": 0.08} + +# ---------------------------------------------------------------- grid helpers +def hold(p, fs, n): + """sample-and-hold of per-frame values onto n 10 ms cells (cell centre decides the frame).""" + idx = ((np.arange(n) + 0.5) * GR / fs).astype(np.int64) + out = np.zeros(n, np.float32) + ok = idx < len(p) + out[ok] = p[idx[ok]] + return out + +@njit(cache=True) +def runs_nb(m): + s = []; e = [] + n = len(m); i = 0 + while i < n: + if m[i]: + j = i + while j < n and m[j]: j += 1 + s.append(i); e.append(j); i = j + else: + i += 1 + return np.array(s, np.int64), np.array(e, np.int64) + +def runs(m): + return runs_nb(np.ascontiguousarray(m, dtype=np.bool_)) + +@njit(cache=True) +def post_nb(m, bridge, min_speech, min_pause, pad, gr): + """repo speech_regions semantics on a grid of period gr: bridge, drop short, merge closer than min_pause, pad (no overlap handling needed + for the mask: padded regions just union).""" + n = len(m); m = m.copy() + # bridge silent gaps between two speech runs + i = 0 + while i < n: + if not m[i]: + j = i + while j < n and not m[j]: j += 1 + if i > 0 and j < n and (j - i) * gr + 1e-9 < bridge: + for k in range(i, j): m[k] = True + i = j + else: + i += 1 + # drop short speech runs + i = 0 + while i < n: + if m[i]: + j = i + while j < n and m[j]: j += 1 + if (j - i) * gr + 1e-9 < min_speech: + for k in range(i, j): m[k] = False + i = j + else: + i += 1 + pf = max(1, int(math.ceil(min_pause / gr - 1e-9))) + out = np.zeros(n, np.bool_) + ra = -1; rb = -1 + pp = int(round(pad / gr)) + i = 0 + while i < n: + if m[i]: + j = i + while j < n and m[j]: j += 1 + if ra >= 0 and i - rb < pf: + rb = j + else: + if ra >= 0: + for k in range(max(0, ra - pp), min(n, rb + pp)): out[k] = True + ra = i; rb = j + i = j + else: + i += 1 + if ra >= 0: + for k in range(max(0, ra - pp), min(n, rb + pp)): out[k] = True + return out + +POST_HEAD = dict(bridge=0.1, min_speech=0.1, min_pause=0.2, pad=0.0) +POST_SIL = dict(bridge=0.1, min_speech=0.25, min_pause=0.1, pad=0.03) + +def post(m, bridge=0.1, min_speech=0.1, min_pause=0.2, pad=0.0): + return post_nb(np.ascontiguousarray(m, dtype=np.bool_), bridge, min_speech, min_pause, pad, GR) + +# ---------------------------------------------------------------- exact native-frame replica of speech_regions +@njit(cache=True) +def native_mask(p, fs, total, thr, bridge, min_speech, min_pause, pad, n10): + """speech_regions() on native frames, returned as a 10 ms bool grid (n10 cells) plus the regions.""" + n = max(0, int(math.ceil(total / fs - 1e-9))) + sp = np.zeros(n, np.bool_) + for f in range(min(n, len(p))): sp[f] = p[f] >= thr + i = 0 + while i < n: + if not sp[i]: + j = i + while j < n and not sp[j]: j += 1 + if i > 0 and j < n and (j - i) * fs + 1e-9 < bridge: + for k in range(i, j): sp[k] = True + i = j + else: + i += 1 + i = 0 + while i < n: + if sp[i]: + j = i + while j < n and sp[j]: j += 1 + if (j - i) * fs + 1e-9 < min_speech: + for k in range(i, j): sp[k] = False + i = j + else: + i += 1 + pf = max(1, int(math.ceil(min_pause / fs - 1e-9))) + rs = []; re_ = [] + i = 0 + while i < n: + if sp[i]: + j = i + while j < n and sp[j]: j += 1 + if len(rs) > 0 and i - re_[-1] < pf: + re_[-1] = j + else: + rs.append(i); re_.append(j) + i = j + else: + i += 1 + R = len(rs) + st = np.zeros(R); en = np.zeros(R) + for r in range(R): + st[r] = rs[r] * fs; en[r] = min(re_[r] * fs, total) + if pad > 0.0: + st2 = st.copy(); en2 = en.copy() + for r in range(R): + st2[r] = max(0.0, st[r] - pad); en2[r] = min(total, en[r] + pad) + for r in range(R - 1): + gap = st[r + 1] - en[r] + if gap < 2.0 * pad: + mid = en[r] + gap / 2.0 + en2[r] = mid; st2[r + 1] = mid + st = st2; en = en2 + out = np.zeros(n10, np.bool_) + for r in range(R): + a = int(round(st[r] / GR)); b = int(round(en[r] / GR)) + for k in range(max(0, a), min(n10, b)): out[k] = True + return out, st, en + +# ---------------------------------------------------------------- fusion / gate +@njit(cache=True) +def two_stage_nb(S, H, pre, post_, Y): + """Silero decides; head extends boundaries outward (pre cells before a start, post_ cells after an end) only where the head says speech + contiguously; fills Silero gaps shorter than Y cells when the head calls >= 50 percent of the gap speech. (rules.py of the fusion study)""" + n = len(S); out = S.copy() + if pre > 0 or post_ > 0: + i = 0 + while i < n: + if S[i]: + j = i + while j < n and S[j]: j += 1 + a = i; b = j + k = a + while k > 0 and a - k < pre and H[k - 1]: k -= 1 + for q in range(k, a): out[q] = True + k = b + while k < n and k - b < post_ and H[k]: k += 1 + for q in range(b, k): out[q] = True + i = j + else: + i += 1 + if Y > 0: + i = 0 + while i < n: + if not out[i]: + j = i + while j < n and not out[j]: j += 1 + if i > 0 and j < n and (j - i) < Y: + c = 0 + for q in range(i, j): + if H[q]: c += 1 + if c * 2 >= (j - i): # mean >= 0.5 + for q in range(i, j): out[q] = True + i = j + else: + i += 1 + return out + +def two_stage(S, H, pre, post_, Y): + return two_stage_nb(np.ascontiguousarray(S, np.bool_), np.ascontiguousarray(H, np.bool_), int(pre), int(post_), int(Y)) + +def gate_frames(p, thr=0.5, med=0.92): + """head gate on native frames: keep a run of p >= thr only if the median probability of its frames is >= med.""" + m = p >= thr + keep = np.zeros(len(p), bool) + s, e = runs(m) + for a, b in zip(s, e): + if np.median(p[a:b]) >= med: keep[a:b] = True + return keep + +# ---------------------------------------------------------------- segmenter replica (segment_by_vad) +LAST_CUTS = [] +def segment_by_mask(sp_in, fs, total, max_seg=30.0, min_seg=1.0, min_pause=0.2, trim=0.3, policy="base", long_gap=None): + """pk::segment_by_vad on an already smoothed boolean speech mask of period fs (the mask is NOT re-smoothed here). + policy: base = the shipped rule (last pause fully inside the window, then pause midpoint in window, then hard cut); + longest = cut at the longest pause in the window; soft = base, but with no pause use the longest silent gap of any length before a hard cut. + long_gap: if set (seconds), additionally cut at every pause at least this long (segments may be shorter than min_seg then).""" + if total <= max_seg: return [(0.0, total)], 0 + n = max(0, int(math.ceil(total / fs - 1e-9))) + sp = np.zeros(n, bool); k = min(n, len(sp_in)); sp[:k] = sp_in[:k] + max_f = max(2, int(math.floor(max_seg / fs + 1e-9))) + min_f = min(max_f - 1, max(1, int(math.ceil(min_seg / fs - 1e-9)))) + pause_f = max(1, int(math.ceil(min_pause / fs - 1e-9))) + sil_s, sil_e = runs(~sp) + pauses = [(a, b) for a, b in zip(sil_s, sil_e) if b - a >= pause_f] + cum = np.concatenate([[0], np.cumsum(sp)]) + out = [] + hard = 0 + cuts = [] + LAST_CUTS.clear() + def emit(a, b, end_sec): + be = min(b, n) + if not (be > a and cum[be] - cum[a] > 0): return + s0, e0 = a * fs, end_sec + if trim > 0: + fa, fb = a, be + nz = np.flatnonzero(sp[a:be]) + fa = a + nz[0]; fb = a + nz[-1] + 1 + s0 = max(s0, fa * fs - trim); e0 = min(e0, fb * fs + trim) + out.append((s0, e0)) + s = 0 + lg = None if long_gap is None else max(1, int(math.ceil(long_gap / fs - 1e-9))) + while total - s * fs > max_seg + 1e-9: + lo, hi = s + min_f, s + max_f + c = -1; c_mid = -1 + if policy == "longest": + best = -1 + for a, b in pauses: + mid = (a + b) // 2 + if a >= lo and b <= hi and b - a > best: best = b - a; c = mid + if c < 0: + for a, b in pauses: + mid = (a + b) // 2 + if lo <= mid <= hi: c_mid = mid + else: + for a, b in pauses: + mid = (a + b) // 2 + if a >= lo and b <= hi: c = mid + if lo <= mid <= hi: c_mid = mid + if c < 0: c = c_mid + if c < 0 and policy == "soft": + best = 0 + for a, b in zip(sil_s, sil_e): + aa, bb = max(a, lo), min(b, hi) + if bb - aa > best: best = bb - aa; c = (aa + bb) // 2 + is_hard = False + if c <= s: + c = hi; hard += 1; is_hard = True + cuts.append((c * fs, is_hard)); LAST_CUTS.append((c * fs, is_hard)) + emit(s, c, c * fs); s = c + emit(s, n, total) + if lg is not None: # split every resulting segment at pauses >= long_gap (inside it), then trim again + out2 = [] + for (a0, b0) in out: + fa = int(round(a0 / fs)); fb = int(round(b0 / fs)) + seg = sp[fa:fb] + ss, ee = runs(~seg) + cuts = [fa + (x + y) // 2 for x, y in zip(ss, ee) if y - x >= lg and x > 0 and y < len(seg)] + pos = fa + for c in cuts + [fb]: + nz = np.flatnonzero(sp[pos:c]) + if len(nz): + out2.append((max(pos * fs, (pos + nz[0]) * fs - trim), min(c * fs, (pos + nz[-1] + 1) * fs + trim))) + pos = c + out = out2 + return out, hard + +def seg_mask(segs, n10): + m = np.zeros(n10, bool) + for a, b in segs: m[int(round(a / GR)):int(round(b / GR))] = True + return m + +# ---------------------------------------------------------------- metrics +def counts(p, r): + n = min(len(p), len(r)); p = p[:n]; r = r[:n] + return np.array([(p & r).sum(), (p & ~r).sum(), (~p & r).sum(), (~p & ~r).sum()], np.int64) + +def prf(c): + tp, fp, fn, _ = c + P = tp / max(1, tp + fp); R = tp / max(1, tp + fn) + return P, R, 2 * P * R / max(1e-12, P + R) + +# ---------------------------------------------------------------- data +def ref_mask(turns_s, turns_e, dur, merge_gap=0.1): + n = int(round(dur / GR)); m = np.zeros(n, bool) + iv = sorted(zip(turns_s, turns_e)) + mg = [] + for s, e in iv: + if mg and s - mg[-1][1] < merge_gap: mg[-1][1] = max(mg[-1][1], e) + else: mg.append([s, e]) + for s, e in mg: m[int(round(s / GR)):int(round(e / GR))] = True + return m + +def split_of(rid, domain): + """recording-disjoint tune/held split. vox: dev=tune, test=held (different videos). ami: validation=tune, test=held (different meetings). + others: md5 parity of the id.""" + if domain in ("vox", "ami"): + return "tune" if rid.startswith(("dev", "validation")) else "held" + return "tune" if int(hashlib.md5(rid.encode()).hexdigest(), 16) % 2 == 0 else "held" + +def load_recordings(): + """returns list of dicts: id, domain, split, dur, n, wav, ref (bool or None for non-speech-only: all False), nonspeech(bool)""" + D = [] + for dom in ("vox", "ami"): + for j in sorted(glob.glob(f"{ROOT}/data/{dom}/*.json")): + m = json.load(open(j)); rid = m["id"] + D.append(dict(id=f"{dom}_{rid}", domain=dom, split=split_of(rid, dom), dur=m["dur"], wav=j[:-5] + ".wav", + ref=ref_mask(m["start"], m["end"], m["dur"]), nonspeech=False)) + for j in sorted(glob.glob(f"{ROOT}/data/ava/*.json")): + m = json.load(open(j)); rid = os.path.basename(j)[:-5] + import soundfile as sf + dur = sf.info(j[:-5] + ".wav").duration + D.append(dict(id=f"ava_{rid}", domain="ava", split=split_of(rid, "ava"), dur=dur, wav=j[:-5] + ".wav", + ref=ref_mask(m["onset"], m["offset"], dur), nonspeech=False)) + import soundfile as sf + for dom, pat in (("music", "data/musan/music_*.wav"), ("noise", "data/musan/noise_*.wav"), ("esc", "data/esc/esc_*.wav")): + for w in sorted(glob.glob(f"{ROOT}/{pat}")): + rid = os.path.basename(w)[:-4]; dur = sf.info(w).duration + D.append(dict(id=f"{dom}_{rid}", domain=dom, split=split_of(rid, dom), dur=dur, wav=w, + ref=np.zeros(int(round(dur / GR)), bool), nonspeech=True)) + return D + +def load_probs(rid): + return {k: np.load(f"{ROOT}/probs/{rid}.{k}.npy") for k in FS} + +def boot_idx(n, B, seed=0): + rng = np.random.default_rng(seed) + return rng.integers(0, n, size=(B, n)) + +@njit(cache=True) +def smooth_native(p, fs, total, thr, bridge, min_speech): + """the speech mask of segment_by_vad after bridging and dropping short runs (no merging at min_pause, no pad), native frames.""" + n = max(0, int(math.ceil(total / fs - 1e-9))) + sp = np.zeros(n, np.bool_) + for f in range(min(n, len(p))): sp[f] = p[f] >= thr + i = 0 + while i < n: + if not sp[i]: + j = i + while j < n and not sp[j]: j += 1 + if i > 0 and j < n and (j - i) * fs + 1e-9 < bridge: + for k in range(i, j): sp[k] = True + i = j + else: + i += 1 + i = 0 + while i < n: + if sp[i]: + j = i + while j < n and sp[j]: j += 1 + if (j - i) * fs + 1e-9 < min_speech: + for k in range(i, j): sp[k] = False + i = j + else: + i += 1 + return sp diff --git a/scripts/vad_bench/real_recordings/wer_dec.py b/scripts/vad_bench/real_recordings/wer_dec.py new file mode 100644 index 0000000..3accfe2 --- /dev/null +++ b/scripts/vad_bench/real_recordings/wer_dec.py @@ -0,0 +1,40 @@ +#!/usr/bin/env python3 +"""WER part: decode explicit segment lists through the throwaway-patched CLI (PK_SEGMENTS), cache decoded words per (asr model, talk, segment). +usage: wer_dec.py plan = {asr: {talk: {system: [[s,e],...]}}} (made by wer_plan.py)""" +import json, os, re, subprocess, sys, tempfile +ROOT = os.environ.get("VAD_REAL_ROOT") or os.getcwd() +MODELS = os.environ.get("VAD_REAL_MODELS", f"{ROOT}/models") +CLI = os.environ.get("PARAKEET_CLI_PATCHED", f"{ROOT}/parakeet-cli-patched") +MODEL = {"ultra": f"{MODELS}/ultra-q8_0.gguf", "redux": f"{MODELS}/redux-packed.gguf"} +def key(s, e): return f"{s:.4f},{e:.4f}" +SET = "ted" +def cpath(asr, talk): return f"{ROOT}/results/wercache/{asr}/{talk}.json" if SET == "ted" else f"{ROOT}/results/wercache_{SET}/{asr}/{talk}.json" +def load(asr, talk): + f = cpath(asr, talk); return json.load(open(f)) if os.path.exists(f) else {} +def run(asr, talk, segs, threads): + cache = load(asr, talk) + miss = sorted({(round(s, 4), round(e, 4)) for s, e in segs if key(s, e) not in cache}) + if not miss: return + with tempfile.TemporaryDirectory(dir=f"{ROOT}/results") as td: + sf, so = f"{td}/seg.txt", f"{td}/out.txt" + open(sf, "w").write("".join(f"{s:.4f} {e:.4f}\n" for s, e in miss)) + r = subprocess.run([CLI, "transcribe", "--model", MODEL[asr], "--input", f"{ROOT}/data/{SET}/{talk}.wav", "--json", "--threads", str(threads), "--vad"], + capture_output=True, text=True, env=dict(os.environ, PK_SEGMENTS=sf, PK_SEGOUT=so)) + if r.returncode != 0: raise RuntimeError(r.stderr[-300:]) + lines = [l for l in open(so).read().split("\n") if l] if os.path.exists(so) else [] + assert len(lines) == len(miss), (len(lines), len(miss), talk) + for (s, e), l in zip(miss, lines): + a, b, w = (l.split("\t") + [""])[:3]; words = [] + for tok in re.split(r" (?=-?\d+\.\d{3}\|-?\d+\.\d{3}\|\d\.\d{3}\|)", w): + if tok: + st, en, cf, tx = tok.split("|", 3); words.append([float(st), float(en), float(cf), tx]) + cache[key(s, e)] = dict(out=[float(a), float(b)], words=words) + os.makedirs(os.path.dirname(cpath(asr, talk)), exist_ok=True) + json.dump(cache, open(cpath(asr, talk) + ".tmp", "w")); os.rename(cpath(asr, talk) + ".tmp", cpath(asr, talk)) +if __name__ == "__main__": + plan = json.load(open(sys.argv[1])); threads = int(sys.argv[2]) if len(sys.argv) > 2 else 8 + SET = plan.pop("_set", "ted") + for asr, talks in plan.items(): + for talk, systems in talks.items(): + allsegs = [tuple(x) for segs in systems.values() for x in segs] + run(asr, talk, allsegs, threads); print("decoded", asr, talk, len(set(allsegs)), flush=True) diff --git a/scripts/vad_bench/real_recordings/wer_plan.py b/scripts/vad_bench/real_recordings/wer_plan.py new file mode 100644 index 0000000..6beb590 --- /dev/null +++ b/scripts/vad_bench/real_recordings/wer_plan.py @@ -0,0 +1,43 @@ +#!/usr/bin/env python3 +"""Build segment lists for the WER experiment from probability dumps (replica of segment_by_vad, trim .3), check the three native ones against the CLI, write results/wer_plan.json.""" +import json, subprocess, sys, os, numpy as np +import vr, systems +SET = sys.argv[1] if len(sys.argv) > 1 else "ted" +TALKS = ["GaryFlake-merged", "RobertGupta-merged", "EricMead_2009P_EricMead-merged", "DanBarber-merged", "MichaelSpecter-merged"] if SET == "ted" else ["GaryFlake+ins", "RobertGupta+ins", "EricMead_2009P_EricMead+ins"] +MOD = {"silero": "silero-vad-f16", "ultra": "ultra-vad-q8_0", "redux": "redux-vad"} +class R: # minimal recording + pass +def rec(t): + import soundfile as sf + dur = sf.info(f"{vr.ROOT}/data/{SET}/{t}.wav").duration + meta = dict(id=("ted_" + t.replace("-merged", "")) if SET == "ted" else ("comp_" + t), ref=np.zeros(int(round(dur / vr.GR)), bool), dur=dur) + return systems.Rec(meta), dur +plan = {"ultra": {}, "redux": {}}; bad = 0; chk = 0 +if SET != "ted": + import subprocess as sp_ + for t in TALKS: + for k in MOD: + if os.path.exists(f"{vr.ROOT}/probs/comp_{t}.{k}.npy"): continue + j = json.loads(sp_.run([vr.CLI_CLEAN, "vad", "--model", f"{vr.MODELS}/{MOD[k]}.gguf", "--input", f"{vr.ROOT}/data/{SET}/{t}.wav", "--probabilities", "--threads", "4"], capture_output=True, text=True).stdout) + np.save(f"{vr.ROOT}/probs/comp_{t}.{k}.npy", np.array(j["probabilities"], np.float32)) +for t in TALKS: + r, dur = rec(t) + S = vr.smooth_native(r.P["silero"], vr.FS["silero"], dur, 0.5, 0.1, 0.25) + seg = {"silero": vr.segment_by_mask(S, vr.FS["silero"], dur, min_pause=0.1)[0]} + for h in ("ultra", "redux"): + H = vr.smooth_native(r.P[h], vr.FS[h], dur, 0.5, 0.1, 0.1) + seg[h] = vr.segment_by_mask(H, vr.FS[h], dur, min_pause=0.2)[0] + for name, spec in (("fusion_def", ("two", h, 0.5, 0.5, 16, 16, 30)), ("fusion_tuned", ("two", h, 0.1, 0.3, 32, 8, 30)), ("gate92", ("gate", h, 0.92)), ("orgate92", ("org", h, 0.92))): + seg[f"{name}_{h}"] = vr.segment_by_mask(systems.mask(r, spec), vr.GR, dur, min_pause=0.2)[0] + # CLI check of the three native segmentations + for k in MOD: + out = json.loads(subprocess.run([vr.CLI_CLEAN, "vad", "--model", f"{vr.MODELS}/{MOD[k]}.gguf", "--input", f"{vr.ROOT}/data/{SET}/{t}.wav", "--mode", "segments", "--threads", "4"], capture_output=True, text=True).stdout) + cli = [(s["start"], s["end"]) for s in out["segments"]]; mine = seg[k]; chk += 1 + ok = len(cli) == len(mine) and all(abs(a - c) < 1.1e-3 and abs(b - d) < 1.1e-3 for (a, b), (c, d) in zip(cli, mine)) + bad += not ok; print(t, k, "cli segs", len(cli), "replica", len(mine), "match", ok, flush=True) + for h in ("ultra", "redux"): + plan[h][t] = {"head": seg[h], "silero": seg["silero"], **{n: seg[f"{n}_{h}"] for n in ("fusion_def", "fusion_tuned", "gate92", "orgate92")}} + plan[h][t] = {k: [[round(a, 4), round(b, 4)] for a, b in v] for k, v in plan[h][t].items()} +print("CLI vs replica: checked", chk, "mismatch", bad) +plan["_set"] = SET +json.dump(plan, open(f"{vr.ROOT}/results/wer_plan.json" if SET == "ted" else f"{vr.ROOT}/results/wer_plan_{SET}.json", "w")) diff --git a/scripts/vad_bench/real_recordings/wer_plan_forced.py b/scripts/vad_bench/real_recordings/wer_plan_forced.py new file mode 100644 index 0000000..47e4220 --- /dev/null +++ b/scripts/vad_bench/real_recordings/wer_plan_forced.py @@ -0,0 +1,14 @@ +#!/usr/bin/env python3 +"""Cost of a mid-speech cut: Silero speech mask cut at pauses with max segment 20 s (pause20) versus cut every 20 s regardless of pauses (forced20).""" +import json, numpy as np, soundfile as sf, vr +TALKS = ["GaryFlake-merged", "RobertGupta-merged", "EricMead_2009P_EricMead-merged", "DanBarber-merged", "MichaelSpecter-merged"] +plan = {"ultra": {}, "redux": {}}; stats = [] +for t in TALKS: + dur = sf.info(f"{vr.ROOT}/data/ted/{t}.wav").duration; P = np.load(f"{vr.ROOT}/probs/ted_{t.replace('-merged','')}.silero.npy") + sp = vr.smooth_native(P, 0.032, dur, 0.5, 0.1, 0.25) + a, ha = vr.segment_by_mask(sp, 0.032, dur, max_seg=20.0, min_pause=0.1) + b, hb = vr.segment_by_mask(sp, 0.032, dur, max_seg=20.0, min_pause=1e6) + stats.append((t, len(a), ha, len(b), hb)) + for h in plan: plan[h][t] = {"pause20": [[round(x, 4), round(y, 4)] for x, y in a], "forced20": [[round(x, 4), round(y, 4)] for x, y in b]} +print(stats); plan["_set"] = "ted" +json.dump(plan, open(f"{vr.ROOT}/results/wer_plan_forced.json", "w")) diff --git a/scripts/vad_bench/real_recordings/wer_report.py b/scripts/vad_bench/real_recordings/wer_report.py new file mode 100644 index 0000000..b10d23d --- /dev/null +++ b/scripts/vad_bench/real_recordings/wer_report.py @@ -0,0 +1,60 @@ +#!/usr/bin/env python3 +"""WER of transcribe --vad style decoding per segmentation system, paired block bootstrap (100 reference words per block, resampled over all talks).""" +import json, os, re, sys, numpy as np, jiwer +ROOT = os.environ.get("VAD_REAL_ROOT") or os.getcwd() +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +import vr +MODE = sys.argv[1] if len(sys.argv) > 1 else "ted" +plan = json.load(open(f"{ROOT}/results/wer_plan.json" if MODE == "ted" else f"{ROOT}/results/wer_plan_{MODE}.json")); SET = plan.pop("_set", "ted") +CD = "wercache" if SET == "ted" else f"wercache_{SET}" +REFD = "ted" if MODE in ("ted", "forced") else "comp" +def norm(s): return re.sub(r"[^a-z0-9' ]", " ", s.lower().replace("-", " ")).split() +def key(s, e): return f"{s:.4f},{e:.4f}" +def hyp(asr, talk, segs): + cache = json.load(open(f"{ROOT}/results/{CD}/{asr}/{talk}.json")); toks = [] + for s, e in segs: + for w in cache[key(s, e)]["words"]: toks += norm(w[3]) + return toks +BL = 100 +def errs(ref, h): + """per-block [errors, ref words], plus S D I""" + nb = (len(ref) + BL - 1) // BL; blk = np.zeros((nb, 2), np.int64); sdi = np.zeros(3, np.int64) + for i in range(len(ref)): blk[i // BL, 1] += 1 + if not h: blk[:, 0] = blk[:, 1]; sdi[1] = len(ref); return blk, sdi + r = jiwer.process_words(" ".join(ref), " ".join(h)); last = 0 + for ch in r.alignments[0]: + if ch.type == "equal": last = ch.ref_end_idx - 1; continue + if ch.type in ("substitute", "delete"): + for k in range(ch.ref_start_idx, ch.ref_end_idx): + blk[min(k, len(ref) - 1) // BL, 0] += 1; last = k + sdi[0 if ch.type == "substitute" else 1] += ch.ref_end_idx - ch.ref_start_idx + else: + n = ch.hyp_end_idx - ch.hyp_start_idx; blk[min(max(ch.ref_start_idx - 1, 0), len(ref) - 1) // BL, 0] += n; sdi[2] += n + return blk, sdi +out = [f"# WER ({MODE}), decoding the segments of each system", ""] +SYS = ["head", "silero", "fusion_def", "fusion_tuned", "gate92", "orgate92"] if MODE != "forced" else ["pause20", "forced20"] +rng = np.random.default_rng(5) +for asr in ("ultra", "redux"): + talks = list(plan[asr].keys()) + if not all(os.path.exists(f"{ROOT}/results/{CD}/{asr}/{t}.json") for t in talks): continue + R = {}; blocks = {}; sdis = {}; dec = {} + for sname in SYS: + allb = []; sd = np.zeros(3, np.int64); secs = 0.0; per = [] + for t in talks: + ref = norm(open(f"{ROOT}/data/{REFD}/{t}.txt").read()); segs = plan[asr][t][sname] + b, s = errs(ref, hyp(asr, t, segs)); allb.append(b); sd += s; secs += sum(e - a for a, e in segs); per.append(b[:, 0].sum() / b[:, 1].sum()) + blocks[sname] = np.concatenate(allb); sdis[sname] = sd; dec[sname] = (secs, per) + B0 = SYS[0]; B1 = "silero" if MODE != "forced" else "forced20" + nb = len(blocks[B0]); bi = rng.integers(0, nb, size=(3000, nb)) + def wer(b, idx=None): + x = b if idx is None else b[idx]; return x[..., 0].sum(-1) / x[..., 1].sum(-1) + out += [f"## {MODE}: {asr} ASR model (VAD segmentation from {asr} head / Silero / fusion / gate)", "", f"| system | WER % [95% CI] | S | D | I | decoded s | per-talk WER % | dWER vs {B0} (pp) | dWER vs {B1} (pp) |", "|---|---|---|---|---|---|---|---|---|"] + for sname in SYS: + b = blocks[sname]; w = wer(b); ws = wer(b, bi); l, h = np.percentile(ws, [2.5, 97.5]) + d1 = ws - wer(blocks[B0], bi); d2 = ws - wer(blocks[B1], bi) + c1 = f"{100*(w-wer(blocks[B0])):+.2f} [{100*np.percentile(d1,2.5):+.2f}, {100*np.percentile(d1,97.5):+.2f}]" + c2 = f"{100*(w-wer(blocks[B1])):+.2f} [{100*np.percentile(d2,2.5):+.2f}, {100*np.percentile(d2,97.5):+.2f}]" + out.append(f"| {sname} | {100*w:.2f} [{100*l:.2f}, {100*h:.2f}] | {sdis[sname][0]} | {sdis[sname][1]} | {sdis[sname][2]} | {dec[sname][0]:.0f} | " + ", ".join(f"{100*x:.1f}" for x in dec[sname][1]) + f" | {c1} | {c2} |") + out.append("") +out.append("talk order: " + ", ".join(plan['ultra'].keys())) +open(f"{ROOT}/results/wer.md" if MODE == "ted" else f"{ROOT}/results/wer_{MODE}.md", "w").write("\n".join(out) + "\n"); print("\n".join(out))