diff --git a/docs/PORTING.md b/docs/PORTING.md index bd231b542..65ae31660 100644 --- a/docs/PORTING.md +++ b/docs/PORTING.md @@ -85,11 +85,9 @@ binary at `GLIBC_2.27`. Prep also measured a host GNU cargo-built helper binary at `GLIBC_2.34`; that measured regression is why host GNU builds are forbidden for the helper release lanes. -| Helper target | Status | Evidence | -|---------------|--------|----------| -| Linux x86_64 glibc | Proven here. | Build, content check, install into a bare venv, and real-inference smoke using the shipped `pyannote-segmentation-3.0.onnx` and `wespeaker-resnet34-256.onnx` assets. | -| Linux aarch64 glibc | Build, cross-link, install, and real-inference smoke are proven on real aarch64 hardware in the release loop. | Local zig GNU cross-link artifact plus real-hardware install/smoke evidence from the release loop. Do not provision an emulator for this lane. | -| macOS arm64 | Built on the macOS build host and executed on the macOS proof host. | macOS build-host wheel, signing/notarization records for the executable and bundled dylib, RECORD repair, and macOS install/smoke proof evidence. This Linux host claims no macOS runtime proof. | +| Helper coverage | Status | Evidence | +|-----------------|--------|----------| +| `solstone/think/probe.py:SOLSTONE_CORE_SPEAKERS_ANALYZE_COVERED_PLATFORMS` | Platform coverage authority. | Build, content check, install into a bare venv, and real-inference smoke using the shipped `pyannote-segmentation-3.0.onnx` and `wespeaker-resnet34-256.onnx` assets for each covered helper platform. Linux helper lanes use zig GNU cross-link artifacts; macOS evidence is produced by the macOS build/proof hosts. Do not provision an emulator for the aarch64 Linux lane. | | Evidence | Repository command | Class | Notes | |----------|--------------------|-------|-------| diff --git a/solstone/observe/sense.py b/solstone/observe/sense.py index 400f3751a..583fd12c2 100644 --- a/solstone/observe/sense.py +++ b/solstone/observe/sense.py @@ -1417,7 +1417,7 @@ def main(): journal_path=journal ) except Exception as exc: - message = f"Speakers-analyze installation is incomplete: {exc}" + message = str(exc) logger.error(message) print(message, file=sys.stderr) raise SystemExit(78) from exc diff --git a/solstone/observe/transcribe/speakers_analyze_adapter.py b/solstone/observe/transcribe/speakers_analyze_adapter.py index ec1b11343..8d9c0c10c 100644 --- a/solstone/observe/transcribe/speakers_analyze_adapter.py +++ b/solstone/observe/transcribe/speakers_analyze_adapter.py @@ -41,7 +41,6 @@ REQUEST_SCHEMA = "solstone-speaker-analyze-request-v1" RESPONSE_SCHEMA = "solstone-speaker-analyze-response-v1" ERROR_SCHEMA = "solstone-speaker-analyze-error-v1" PRODUCER_ID = "solstone-core-speakers-analyze-v1" -EXIT_UNAVAILABLE = 69 TEMP_ROOT = Path("/var/tmp") TEMP_PREFIX = "solstone-speakers-analyze-" TEMP_DIR_MODE = 0o700 @@ -149,7 +148,7 @@ def analyze_speakers( pyannote_model_path=pyannote_model_path, ) request_ids = [int(statement["id"]) for statement in statements_pre_restore] - expected_statement_ids = _python_admitted_statement_ids( + expected_statement_ids = _request_admitted_statement_ids( statement_audio, statements_pre_restore, sample_rate=sample_rate, @@ -169,7 +168,6 @@ def analyze_speakers( ) from exc return _accepted_result_from_response( response, - raw_path=raw_path, payload_path=payload_path, statements_restored=restored_statements, expected_statement_ids=expected_statement_ids, @@ -186,10 +184,6 @@ def analyze_speakers( raise SpeakerAnalyzeError( path=raw_path, stage="request", reason=type(exc).__name__.lower() ) from exc - except Exception as exc: - raise SpeakerAnalyzeError( - path=raw_path, stage="request", reason=type(exc).__name__.lower() - ) from exc finally: if temp_dir is not None: shutil.rmtree(temp_dir, ignore_errors=True) @@ -205,6 +199,7 @@ def invoke_speakers_analyze_helper( selector_factory=selectors.DefaultSelector, clock: Callable[[], float] = time.monotonic, ) -> HelperInvocationResult: + deadline = clock() + budget.timeout_s try: proc = popen_factory( argv, @@ -219,32 +214,59 @@ def invoke_speakers_analyze_helper( assert proc.stdin is not None assert proc.stdout is not None assert proc.stderr is not None - try: - proc.stdin.write(stdin_text.encode("utf-8")) - proc.stdin.close() - except BrokenPipeError: - pass + stdin_bytes = memoryview(stdin_text.encode("utf-8")) + stdin_offset = 0 + stdin_open = True stdout = bytearray() stderr = bytearray() - deadline = clock() + budget.timeout_s with selector_factory() as selector: + os.set_blocking(proc.stdin.fileno(), False) os.set_blocking(proc.stdout.fileno(), False) os.set_blocking(proc.stderr.fileno(), False) + if stdin_bytes: + selector.register(proc.stdin, selectors.EVENT_WRITE, "stdin") + else: + proc.stdin.close() + stdin_open = False selector.register(proc.stdout, selectors.EVENT_READ, "stdout") selector.register(proc.stderr, selectors.EVENT_READ, "stderr") while selector.get_map(): remaining = deadline - clock() if remaining <= 0: + reason = ( + "stdin-write-timeout" + if stdin_open and stdin_offset < len(stdin_bytes) + else "timeout" + ) _terminate_and_reap(proc, budget) raise SpeakerAnalyzeError( path=raw_path, stage="invoke", - reason="timeout", + reason=reason, native_exit_code=proc.returncode, ) for key, _events in selector.select(timeout=min(0.1, remaining)): stream_name = key.data + if stream_name == "stdin": + try: + written = os.write( + key.fileobj.fileno(), + stdin_bytes[stdin_offset : stdin_offset + 8192], + ) + except BlockingIOError: + continue + except BrokenPipeError: + selector.unregister(key.fileobj) + key.fileobj.close() + stdin_open = False + continue + stdin_offset += written + if stdin_offset >= len(stdin_bytes): + selector.unregister(key.fileobj) + key.fileobj.close() + stdin_open = False + continue chunk = os.read(key.fileobj.fileno(), 8192) if not chunk: selector.unregister(key.fileobj) @@ -397,7 +419,7 @@ def _ensure_span_parity( raise NativePayloadError("request", "span-parity-statement-id") -def _python_admitted_statement_ids( +def _request_admitted_statement_ids( audio: np.ndarray, statements: list[dict[str, Any]], *, @@ -430,7 +452,6 @@ def _python_admitted_statement_ids( def _accepted_result_from_response( response: object, *, - raw_path: Path, payload_path: Path, statements_restored: list[dict[str, Any]], expected_statement_ids: list[int], @@ -727,7 +748,6 @@ class NativePayloadError(RuntimeError): __all__ = [ "DEFAULT_INVOCATION_BUDGET", - "EXIT_UNAVAILABLE", "PRODUCER_ID", "RESPONSE_SCHEMA", "SpeakerAnalyzeResult", diff --git a/tests/test_sense.py b/tests/test_sense.py index 7751746c7..ab9415cf3 100644 --- a/tests/test_sense.py +++ b/tests/test_sense.py @@ -2868,6 +2868,41 @@ def test_main_rejects_invalid_stream_filter(tmp_path, monkeypatch): assert exc_info.value.code == 2 +def test_main_speakers_analyze_failure_prints_canonical_message_once( + tmp_path, monkeypatch, capsys +): + from solstone.observe import sense + from solstone.think.speakers_analyze_installation import ( + SPEAKERS_ANALYZE_REPAIR_TEXT, + ) + + message = ( + "Speakers-analyze installation is incomplete " + f"(asset-missing: wespeaker). {SPEAKERS_ANALYZE_REPAIR_TEXT}" + ) + + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + monkeypatch.setattr(sense, "require_solstone", lambda: None) + monkeypatch.setattr(sys, "argv", ["sense", "--day", "20250101"]) + + def fail_generation(**_kwargs): + raise RuntimeError(message) + + monkeypatch.setattr( + "solstone.think.speakers_analyze_installation." + "begin_speakers_analyze_generation", + fail_generation, + ) + + with pytest.raises(SystemExit) as exc_info: + sense.main() + + captured = capsys.readouterr() + assert exc_info.value.code == 78 + assert captured.err.count("Speakers-analyze installation is incomplete") == 1 + assert message in captured.err + + def _registered_describe_commands(sensor: FileSensor) -> list[list[str]]: return [ command diff --git a/tests/test_speakers_analyze_adapter.py b/tests/test_speakers_analyze_adapter.py index d62050093..77cd92c65 100644 --- a/tests/test_speakers_analyze_adapter.py +++ b/tests/test_speakers_analyze_adapter.py @@ -289,6 +289,40 @@ def test_payload_size_checked_before_read( assert exc.value.reason == "embedding-payload-size-mismatch" +def test_unexpected_internal_exception_propagates_and_cleans_temp_dir(tmp_path: Path): + class UnexpectedAdapterBug(RuntimeError): + pass + + temp_dir = tmp_path / "adapter-temp" + + def temp_dir_factory(_raw_path: Path) -> Path: + temp_dir.mkdir(mode=0o700) + return temp_dir + + def statements_restored() -> list[dict[str, Any]]: + raise UnexpectedAdapterBug("boom") + + with pytest.raises(UnexpectedAdapterBug, match="boom"): + analyze_speakers( + raw_path=tmp_path / "audio.flac", + full_audio=np.zeros(20, dtype=np.float32), + statement_audio=np.zeros(20, dtype=np.float32), + reduced_audio=None, + statements_pre_restore=[{"id": 1, "start": 0.0, "end": 0.5, "text": "x"}], + statements_restored=statements_restored, + sample_rate=10, + min_statement_duration=0.3, + helper_locator=lambda: tmp_path / "helper", + helper_invoker=lambda _argv, _stdin, _raw_path: pytest.fail( + "helper should not be invoked after restoration fails" + ), + model_path_resolver=lambda: (tmp_path / "w.onnx", tmp_path / "p.onnx"), + temp_dir_factory=temp_dir_factory, + ) + + assert not temp_dir.exists() + + def test_gate_decline_null_labels_is_accepted(tmp_path: Path): result, _request, _temp_dir = _run_adapter( tmp_path, diff --git a/tests/test_speakers_analyze_invocation.py b/tests/test_speakers_analyze_invocation.py index 64cfdf066..9e23126c8 100644 --- a/tests/test_speakers_analyze_invocation.py +++ b/tests/test_speakers_analyze_invocation.py @@ -61,6 +61,49 @@ def test_timeout_terminates_and_reaps_child(tmp_path: Path): assert exc.value.reason == "timeout" +def test_stdin_write_timeout_terminates_reaps_and_cleans_temp_dir(tmp_path: Path): + helper = tmp_path / "never_reads_stdin.py" + helper.write_text( + "#!/usr/bin/env python3\nimport time\ntime.sleep(10)\n", + encoding="utf-8", + ) + helper.chmod(0o755) + temp_dir = tmp_path / "adapter-temp" + statements = [ + {"id": statement_id, "start": 0.0, "end": 1.0, "text": "x"} + for statement_id in range(20_000) + ] + + def temp_dir_factory(_raw_path: Path) -> Path: + temp_dir.mkdir(mode=0o700) + return temp_dir + + with pytest.raises(SpeakerAnalyzeError) as exc: + analyze_speakers( + raw_path=tmp_path / "audio.flac", + full_audio=np.zeros(10, dtype=np.float32), + statement_audio=np.zeros(10, dtype=np.float32), + reduced_audio=None, + statements_pre_restore=statements, + statements_restored=statements, + sample_rate=10, + min_statement_duration=0.3, + helper_locator=lambda: helper, + helper_invoker=lambda argv, stdin, raw_path: invoke_speakers_analyze_helper( + argv, + stdin, + raw_path, + budget=_budget(timeout_s=0.01), + ), + model_path_resolver=lambda: (tmp_path / "w.onnx", tmp_path / "p.onnx"), + temp_dir_factory=temp_dir_factory, + ) + + assert exc.value.stage == "invoke" + assert exc.value.reason == "stdin-write-timeout" + assert not temp_dir.exists() + + @pytest.mark.parametrize( ("stream", "reason"), [("stdout", "stdout-too-large"), ("stderr", "stderr-too-large")], diff --git a/tests/test_transcribe.py b/tests/test_transcribe.py index 0b59ba2e2..00a9f421d 100644 --- a/tests/test_transcribe.py +++ b/tests/test_transcribe.py @@ -884,6 +884,90 @@ def test_all_batch_typed_speaker_failure_continues_and_preserves_failed_audio( assert len(speaker_failure_events) == 1 +def test_all_batch_unexpected_adapter_exception_aborts(tmp_path, monkeypatch): + from solstone.observe.transcribe.main import main + from solstone.think.speakers_analyze_installation import ( + SpeakersAnalyzeInstallationResult, + ) + + class UnexpectedAdapterBug(RuntimeError): + pass + + first = ( + tmp_path / "chronicle" / "20260416" / "default" / "120000_300" / "audio.flac" + ) + second = ( + tmp_path / "chronicle" / "20260416" / "default" / "121000_300" / "audio.flac" + ) + first.parent.mkdir(parents=True) + second.parent.mkdir(parents=True) + first.write_bytes(b"first") + second.write_bytes(b"second") + statements = [{"id": 0, "start": 0.0, "end": 1.0, "text": "hi"}] + backend_module = MagicMock() + backend_module.get_model_info.return_value = { + "model": "medium.en", + "device": "cpu", + "compute_type": "int8", + } + + monkeypatch.setenv("SOLSTONE_JOURNAL", str(tmp_path)) + monkeypatch.setattr(sys, "argv", ["journal transcribe", "--all"]) + with ( + patch( + "solstone.think.speakers_analyze_installation." + "check_speakers_analyze_installation", + return_value=SpeakersAnalyzeInstallationResult("ok"), + ), + patch( + "solstone.observe.transcribe.main.read_available_bytes", + return_value=8 * 1024**3, + ), + patch( + "solstone.observe.transcribe.main.stt_local_floor_bytes", + return_value=4 * 1024**3, + ), + patch( + "solstone.observe.transcribe.main.local_stt_backend", + return_value="parakeet", + ), + patch( + "solstone.observe.transcribe.main.load_audio", + return_value=np.zeros(10 * SAMPLE_RATE, dtype=np.float32), + ), + patch( + "solstone.observe.vad.run_vad", + return_value=VadResult( + duration=10.0, + speech_duration=5.0, + has_speech=True, + speech_segments=[(1.0, 6.0)], + ), + ), + patch("solstone.observe.vad.reduce_audio", return_value=(None, None)), + patch("solstone.observe.transcribe.main.tag_audio", return_value=None), + patch( + "solstone.observe.transcribe.main.stt_transcribe", + return_value=statements, + ), + patch( + "solstone.observe.transcribe.main.get_backend", + return_value=backend_module, + ), + patch( + "solstone.observe.transcribe.speakers_analyze_adapter.analyze_speakers", + side_effect=UnexpectedAdapterBug("boom"), + ), + ): + with pytest.raises(SystemExit) as exc_info: + main() + + assert exc_info.value.code == 1 + assert isinstance(exc_info.value.__cause__, UnexpectedAdapterBug) + assert not first.with_suffix(".jsonl").exists() + assert not second.with_suffix(".jsonl").exists() + + def test_single_file_typed_speaker_failure_emits_once_and_exits_one( tmp_path, monkeypatch: pytest.MonkeyPatch,