diff --git a/docs/JOURNAL.md b/docs/JOURNAL.md index 7ae721bae..24fb77f2b 100644 --- a/docs/JOURNAL.md +++ b/docs/JOURNAL.md @@ -240,6 +240,7 @@ The `transcribe` block configures audio transcription settings for `sol transcri "backend": "whisper", "enrich": true, "preserve_all": false, + "noise_upgrade_min_speech_ratio": 0.3, "whisper": { "device": "auto", "model": "medium.en", @@ -256,6 +257,7 @@ The `transcribe` block configures audio transcription settings for `sol transcri - `backend` (string) – STT backend to use: `"whisper"` (local processing) or `"revai"` (cloud with speaker diarization). Default: `"whisper"`. - `enrich` (boolean) – Enable LLM enrichment for topic extraction and transcript correction. Default: `true`. - `preserve_all` (boolean) – Keep audio files even when no speech is detected. When `false`, silent recordings are deleted to save disk space. Default: `false`. +- `noise_upgrade_min_speech_ratio` (number) – Min speech/loud ratio required for noisy upgrade (default: `0.3`). Filters out music and other non-speech noise. **Whisper backend settings** (`transcribe.whisper`): - `device` (string) – Device for inference: `"auto"` (detect GPU, fall back to CPU), `"cpu"`, or `"cuda"`. Default: `"auto"`. diff --git a/observe/transcribe/main.py b/observe/transcribe/main.py index 7066efa23..2490e6f56 100644 --- a/observe/transcribe/main.py +++ b/observe/transcribe/main.py @@ -21,6 +21,7 @@ Configuration (journal config transcribe section): - transcribe.preserve_all: Keep audio files even when no speech detected (default: false) - transcribe.min_speech_seconds: Minimum speech duration to proceed. Default: 1.0 - transcribe.noise_upgrade: Auto-switch to Rev.ai for noisy recordings (default: true) +- transcribe.noise_upgrade_min_speech_ratio: Min speech/loud ratio required for noisy upgrade (default: 0.3). Filters out music and other non-speech noise. Whisper backend settings (transcribe.whisper): - device: Device for inference ("auto", "cpu", "cuda"). Default: "auto" @@ -205,6 +206,12 @@ def _build_base_event( if vad_result.noisy_rms is not None: event["noisy_rms"] = round(vad_result.noisy_rms, 4) event["noisy_s"] = round(vad_result.noisy_s, 1) + if vad_result.loud_windows > 0: + event["loud_windows"] = vad_result.loud_windows + event["speech_loud_windows"] = vad_result.speech_loud_windows + ratio = vad_result.loud_speech_ratio + if ratio is not None: + event["loud_speech_ratio"] = round(ratio, 2) if day: event["day"] = day @@ -359,6 +366,12 @@ def _statements_to_jsonl( if vad_result.noisy_rms is not None: metadata["noisy_rms"] = round(vad_result.noisy_rms, 4) metadata["noisy_s"] = round(vad_result.noisy_s, 1) + if vad_result.loud_windows > 0: + metadata["loud_windows"] = vad_result.loud_windows + metadata["speech_loud_windows"] = vad_result.speech_loud_windows + ratio = vad_result.loud_speech_ratio + if ratio is not None: + metadata["loud_speech_ratio"] = round(ratio, 2) # Add enrichment metadata if available if enrichment: @@ -505,23 +518,6 @@ def process_audio( backend_module = get_backend(backend) model_info = backend_module.get_model_info(backend_config) - # Sanity check: if VAD detected speech but we got no statements, something is wrong - if vad_result.has_speech and not statements: - if vad_result.speech_duration < 5.0: - # Marginal speech detection — treat as silence rather than failure. - # VAD occasionally flags brief noise as speech; if the STT backend - # can't produce anything from it, that's expected, not an error. - logging.info( - f"VAD detected {vad_result.speech_duration:.1f}s of marginal speech " - f"but transcription produced 0 statements — treating as silence" - ) - else: - raise RuntimeError( - f"VAD detected {vad_result.speech_duration:.1f}s of speech " - f"(from {vad_result.duration:.1f}s total) but transcription produced " - f"0 statements. This indicates a transcription failure, not silence." - ) - # Load config for preserve_all setting config = get_config() preserve_all = config.get("transcribe", {}).get("preserve_all", False) @@ -531,6 +527,12 @@ def process_audio( # Handle no speech detected if not statements: + logging.info( + "STT backend returned 0 statements, treating as silence " + "(VAD: %.1fs speech of %.1fs)", + vad_result.speech_duration, + vad_result.duration, + ) if preserve_all: event["outcome"] = "preserved" logging.info( @@ -706,6 +708,7 @@ def _process_one( # - Audio is noisy # - Rev.ai token is available noise_upgrade = transcribe_config.get("noise_upgrade", True) + min_ratio = transcribe_config.get("noise_upgrade_min_speech_ratio", 0.3) if ( not args.backend and noise_upgrade @@ -714,10 +717,20 @@ def _process_one( ): from observe.transcribe.revai import has_token - if has_token(): + ratio = vad_result.loud_speech_ratio + if ratio is not None and ratio < min_ratio: + logging.info( + "Noisy audio (RMS=%.4f) looks like non-speech (loud_speech_ratio=%.2f < %.2f), " + "skipping Rev.ai upgrade", + vad_result.noisy_rms, + ratio, + min_ratio, + ) + elif has_token(): logging.info( - f"Noisy audio detected (RMS={vad_result.noisy_rms:.4f}), " - f"upgrading to Rev.ai backend" + "Noisy audio detected (RMS=%.4f, loud_speech_ratio=%s), upgrading to Rev.ai backend", + vad_result.noisy_rms, + f"{ratio:.2f}" if ratio is not None else "n/a", ) backend = "revai" diff --git a/observe/vad.py b/observe/vad.py index 67fb5d826..23c1a88d0 100644 --- a/observe/vad.py +++ b/observe/vad.py @@ -113,6 +113,40 @@ def compute_nonspeech_rms( return float(np.mean(rms_values)), total_duration +def compute_loud_speech_windows( + audio: np.ndarray, + speech_segments: list[tuple[float, float]], + sample_rate: int, + window_s: float = 1.0, + rms_threshold: float = 0.01, +) -> tuple[int, int]: + """Count loud fixed-duration windows and those that overlap speech.""" + window_samples = int(window_s * sample_rate) + if len(audio) < window_samples: + return 0, 0 + + loud_windows = 0 + speech_loud_windows = 0 + + for i in range(len(audio) // window_samples): + start_sample = i * window_samples + end_sample = start_sample + window_samples + window_audio = audio[start_sample:end_sample] + rms = np.sqrt(np.mean(window_audio**2)) + if rms <= rms_threshold: + continue + + loud_windows += 1 + window_start = i * window_s + window_end = window_start + window_s + if any( + window_start < end and window_end > start for start, end in speech_segments + ): + speech_loud_windows += 1 + + return loud_windows, speech_loud_windows + + @dataclass class SpeechSegment: """A segment of speech with original and reduced timestamps. @@ -203,6 +237,9 @@ class VadResult: speech_segments: List of (start, end) tuples for each speech segment noisy_rms: RMS level of non-speech regions (None if not computable) noisy_s: Duration of non-speech audio used for RMS calculation + loud_windows: Number of 1s windows whose RMS exceeds the loud threshold + speech_loud_windows: Number of loud windows that overlap speech segments + loud_speech_ratio: Ratio of speech_loud_windows to loud_windows, or None """ duration: float @@ -211,6 +248,8 @@ class VadResult: speech_segments: list[tuple[float, float]] = field(default_factory=list) noisy_rms: float | None = None noisy_s: float = 0.0 + loud_windows: int = 0 + speech_loud_windows: int = 0 def is_noisy(self, threshold: float = 0.01) -> bool: """Check if background noise level exceeds threshold. @@ -230,6 +269,13 @@ class VadResult: return 0.0 return self.speech_duration / self.duration + @property + def loud_speech_ratio(self) -> float | None: + """Ratio of speech-overlapping loud windows to all loud windows.""" + if self.loud_windows <= 0: + return None + return self.speech_loud_windows / self.loud_windows + def run_vad( audio: np.ndarray, @@ -274,13 +320,22 @@ def run_vad( # Compute RMS of non-speech regions (for noise detection) noisy_rms, noisy_s = compute_nonspeech_rms(audio, speech_segments, SAMPLE_RATE) + loud_windows, speech_loud_windows = compute_loud_speech_windows( + audio, speech_segments, SAMPLE_RATE + ) vad_time = time.perf_counter() - t0 rms_str = f", rms={noisy_rms:.4f}" if noisy_rms is not None else "" + ratio = speech_loud_windows / loud_windows if loud_windows > 0 else None + loud_str = ( + f", loud_windows={loud_windows}, speech_loud={speech_loud_windows}, ratio={ratio:.2f}" + if ratio is not None + else "" + ) logging.info( f" VAD complete in {vad_time:.2f}s: " f"{duration:.1f}s total, {speech_duration:.1f}s speech, " - f"{len(speech_chunks)} chunks, has_speech={has_speech}{rms_str}" + f"{len(speech_chunks)} chunks, has_speech={has_speech}{rms_str}{loud_str}" ) return VadResult( @@ -290,6 +345,8 @@ def run_vad( speech_segments=speech_segments, noisy_rms=noisy_rms, noisy_s=noisy_s, + loud_windows=loud_windows, + speech_loud_windows=speech_loud_windows, ) diff --git a/tests/test_transcribe_empty_result.py b/tests/test_transcribe_empty_result.py new file mode 100644 index 000000000..65206dc7d --- /dev/null +++ b/tests/test_transcribe_empty_result.py @@ -0,0 +1,106 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +"""Tests for empty-result handling in process_audio.""" + +from unittest.mock import MagicMock, patch + +import numpy as np +import pytest + +from observe.utils import SAMPLE_RATE +from observe.vad import VadResult + + +@pytest.fixture +def raw_path(tmp_path): + path = tmp_path / "chronicle" / "20260416" / "default" / "120000_300" / "audio.m4a" + path.parent.mkdir(parents=True) + path.touch() + return path + + +@pytest.fixture +def audio_buffer(): + return np.zeros(10 * SAMPLE_RATE, dtype=np.float32) + + +@pytest.fixture +def vad_result(): + return VadResult( + duration=10.0, + speech_duration=5.0, + has_speech=True, + speech_segments=[(1.0, 6.0)], + ) + + +def test_empty_statements_filter_path(raw_path, audio_buffer, vad_result): + from observe.transcribe.main import process_audio + + backend_module = MagicMock() + backend_module.get_model_info.return_value = { + "model": "medium.en", + "device": "cpu", + "compute_type": "int8", + } + + with ( + patch( + "observe.transcribe.main.get_config", + return_value={"transcribe": {"preserve_all": False}}, + ), + patch( + "observe.transcribe.main.get_journal", return_value=str(raw_path.parents[4]) + ), + patch("observe.transcribe.main.stt_transcribe", return_value=[]), + patch("observe.transcribe.main.get_backend", return_value=backend_module), + patch("observe.transcribe.main.callosum_send") as mock_send, + ): + process_audio(raw_path, audio_buffer, vad_result, {}, backend="whisper") + + assert not raw_path.exists() + assert mock_send.call_args.args[:2] == ("observe", "transcribed") + assert mock_send.call_args.kwargs["outcome"] == "filtered" + + +def test_empty_statements_preserve_path(raw_path, audio_buffer, vad_result): + from observe.transcribe.main import process_audio + + backend_module = MagicMock() + backend_module.get_model_info.return_value = { + "model": "medium.en", + "device": "cpu", + "compute_type": "int8", + } + + with ( + patch( + "observe.transcribe.main.get_config", + return_value={"transcribe": {"preserve_all": True}}, + ), + patch( + "observe.transcribe.main.get_journal", return_value=str(raw_path.parents[4]) + ), + patch("observe.transcribe.main.stt_transcribe", return_value=[]), + patch("observe.transcribe.main.get_backend", return_value=backend_module), + patch("observe.transcribe.main.callosum_send") as mock_send, + ): + process_audio(raw_path, audio_buffer, vad_result, {}, backend="whisper") + + assert raw_path.exists() + assert mock_send.call_args.args[:2] == ("observe", "transcribed") + assert mock_send.call_args.kwargs["outcome"] == "preserved" + + +def test_backend_raise_propagates(raw_path, audio_buffer, vad_result): + from observe.transcribe.main import process_audio + + with patch( + "observe.transcribe.main.stt_transcribe", side_effect=RuntimeError("rev.ai 502") + ): + with pytest.raises(SystemExit) as exc_info: + process_audio(raw_path, audio_buffer, vad_result, {}, backend="whisper") + + assert exc_info.value.code == 1 + assert raw_path.exists() diff --git a/tests/test_transcribe_noise_upgrade.py b/tests/test_transcribe_noise_upgrade.py index 0c475f964..e1aa560ac 100644 --- a/tests/test_transcribe_noise_upgrade.py +++ b/tests/test_transcribe_noise_upgrade.py @@ -3,11 +3,34 @@ """Tests for noise upgrade feature in transcription.""" +import argparse from unittest.mock import patch +import numpy as np +import pytest + +from observe.utils import SAMPLE_RATE from observe.vad import VadResult +@pytest.fixture +def audio_path(tmp_path): + path = tmp_path / "chronicle" / "20260416" / "default" / "120000_300" / "audio.m4a" + path.parent.mkdir(parents=True) + path.touch() + return path + + +@pytest.fixture +def args(): + return argparse.Namespace(backend=None, cpu=False, model=None, redo=False) + + +@pytest.fixture +def audio_buffer(): + return np.zeros(10 * SAMPLE_RATE, dtype=np.float32) + + class TestHasToken: """Tests for revai.has_token() function.""" @@ -171,3 +194,129 @@ class TestBackendMetadata: lines = _statements_to_jsonl(statements, "audio.flac", base_dt, model_info) metadata = json.loads(lines[0]) assert metadata["backend"] == "unknown" + + +class TestNoiseUpgradeGate: + def test_gate_blocks_on_low_ratio(self, audio_path, args, audio_buffer): + from observe.transcribe.main import _process_one + + vad = VadResult( + duration=10.0, + speech_duration=5.0, + has_speech=True, + speech_segments=[(1.0, 6.0)], + noisy_rms=0.02, + noisy_s=3.0, + loud_windows=200, + speech_loud_windows=10, + ) + transcribe_config = { + "backend": "whisper", + "noise_upgrade": True, + "noise_upgrade_min_speech_ratio": 0.3, + "whisper": {}, + } + + with ( + patch("observe.transcribe.main.load_audio", return_value=audio_buffer), + patch("observe.transcribe.main.run_vad", return_value=vad), + patch("observe.transcribe.main.reduce_audio", return_value=(None, None)), + patch("observe.transcribe.main.process_audio") as mock_process_audio, + patch("observe.transcribe.revai.has_token", return_value=True), + ): + _process_one(audio_path, args, transcribe_config, []) + + assert mock_process_audio.call_args.kwargs["backend"] == "whisper" + + def test_gate_admits_on_high_ratio(self, audio_path, args, audio_buffer): + from observe.transcribe.main import _process_one + + vad = VadResult( + duration=10.0, + speech_duration=5.0, + has_speech=True, + speech_segments=[(1.0, 6.0)], + noisy_rms=0.02, + noisy_s=3.0, + loud_windows=100, + speech_loud_windows=90, + ) + transcribe_config = { + "backend": "whisper", + "noise_upgrade": True, + "noise_upgrade_min_speech_ratio": 0.3, + "whisper": {}, + } + + with ( + patch("observe.transcribe.main.load_audio", return_value=audio_buffer), + patch("observe.transcribe.main.run_vad", return_value=vad), + patch("observe.transcribe.main.reduce_audio", return_value=(None, None)), + patch("observe.transcribe.main.process_audio") as mock_process_audio, + patch("observe.transcribe.revai.has_token", return_value=True), + ): + _process_one(audio_path, args, transcribe_config, []) + + assert mock_process_audio.call_args.kwargs["backend"] == "revai" + + def test_gate_fallback_when_ratio_none(self, audio_path, args, audio_buffer): + from observe.transcribe.main import _process_one + + vad = VadResult( + duration=10.0, + speech_duration=5.0, + has_speech=True, + speech_segments=[(1.0, 6.0)], + noisy_rms=0.02, + noisy_s=3.0, + loud_windows=0, + speech_loud_windows=0, + ) + transcribe_config = { + "backend": "whisper", + "noise_upgrade": True, + "noise_upgrade_min_speech_ratio": 0.3, + "whisper": {}, + } + + with ( + patch("observe.transcribe.main.load_audio", return_value=audio_buffer), + patch("observe.transcribe.main.run_vad", return_value=vad), + patch("observe.transcribe.main.reduce_audio", return_value=(None, None)), + patch("observe.transcribe.main.process_audio") as mock_process_audio, + patch("observe.transcribe.revai.has_token", return_value=True), + ): + _process_one(audio_path, args, transcribe_config, []) + + assert mock_process_audio.call_args.kwargs["backend"] == "revai" + + def test_gate_blocks_when_not_noisy(self, audio_path, args, audio_buffer): + from observe.transcribe.main import _process_one + + vad = VadResult( + duration=10.0, + speech_duration=5.0, + has_speech=True, + speech_segments=[(1.0, 6.0)], + noisy_rms=0.005, + noisy_s=3.0, + loud_windows=100, + speech_loud_windows=90, + ) + transcribe_config = { + "backend": "whisper", + "noise_upgrade": True, + "noise_upgrade_min_speech_ratio": 0.3, + "whisper": {}, + } + + with ( + patch("observe.transcribe.main.load_audio", return_value=audio_buffer), + patch("observe.transcribe.main.run_vad", return_value=vad), + patch("observe.transcribe.main.reduce_audio", return_value=(None, None)), + patch("observe.transcribe.main.process_audio") as mock_process_audio, + patch("observe.transcribe.revai.has_token", return_value=True), + ): + _process_one(audio_path, args, transcribe_config, []) + + assert mock_process_audio.call_args.kwargs["backend"] == "whisper" diff --git a/tests/test_vad.py b/tests/test_vad.py index 540012888..697a1ac6d 100644 --- a/tests/test_vad.py +++ b/tests/test_vad.py @@ -6,6 +6,7 @@ from unittest.mock import patch import numpy as np +import pytest from observe.utils import SAMPLE_RATE from observe.vad import ( @@ -13,6 +14,7 @@ from observe.vad import ( AudioReduction, SpeechSegment, VadResult, + compute_loud_speech_windows, compute_nonspeech_rms, get_nonspeech_segments, reduce_audio, @@ -21,6 +23,11 @@ from observe.vad import ( ) +@pytest.fixture +def rng(): + return np.random.default_rng(42) + + class TestVadResult: """Test VadResult dataclass.""" @@ -37,6 +44,9 @@ class TestVadResult: assert result.speech_duration == 5.0 assert result.has_speech is True assert result.speech_segments == [(1.0, 3.0), (5.0, 8.0)] + assert result.loud_windows == 0 + assert result.speech_loud_windows == 0 + assert result.loud_speech_ratio is None def test_vad_result_no_speech(self): """VadResult with no speech should have has_speech=False.""" @@ -402,6 +412,57 @@ class TestRunVad: assert result.noisy_rms is None assert result.noisy_s == 0.0 + def test_loud_speech_ratio_music_like_fixture(self, rng): + audio = rng.normal(0.0, 0.1, 30 * SAMPLE_RATE).astype(np.float32) + speech_segments = [(5.0, 6.0), (20.0, 21.0)] + + loud_windows, speech_loud_windows = compute_loud_speech_windows( + audio, speech_segments, SAMPLE_RATE + ) + + assert loud_windows >= 25 + assert speech_loud_windows <= 3 + assert speech_loud_windows / loud_windows < 0.15 + + def test_loud_speech_ratio_meeting_like_fixture(self): + audio = np.zeros(30 * SAMPLE_RATE, dtype=np.float32) + audio[0 : 10 * SAMPLE_RATE] = 0.1 + audio[15 * SAMPLE_RATE : 25 * SAMPLE_RATE] = 0.1 + speech_segments = [(0.0, 10.0), (15.0, 25.0)] + + loud_windows, speech_loud_windows = compute_loud_speech_windows( + audio, speech_segments, SAMPLE_RATE + ) + + assert loud_windows == 20 + assert speech_loud_windows == 20 + assert speech_loud_windows / loud_windows == pytest.approx(1.0, rel=0.01) + + def test_loud_speech_ratio_all_silent(self): + audio = np.zeros(10 * SAMPLE_RATE, dtype=np.float32) + + loud_windows, speech_loud_windows = compute_loud_speech_windows( + audio, [], SAMPLE_RATE + ) + + assert loud_windows == 0 + assert speech_loud_windows == 0 + assert ( + VadResult( + duration=10.0, + speech_duration=0.0, + has_speech=False, + loud_windows=0, + speech_loud_windows=0, + ).loud_speech_ratio + is None + ) + + def test_loud_speech_ratio_short_audio(self): + audio = np.zeros(SAMPLE_RATE // 2, dtype=np.float32) + + assert compute_loud_speech_windows(audio, [], SAMPLE_RATE) == (0, 0) + class TestSpeechSegment: """Test SpeechSegment dataclass."""