From fd46697c10b6ebd15d51b5423716ffc94743cf1f Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Sat, 11 Jul 2026 06:17:04 -0600 Subject: [PATCH] perf(observe): cap local Qwen categorization images --- docs/OBSERVE.md | 20 ++++++++++++++- solstone/observe/describe.py | 23 ++++++++++++++++- solstone/observe/utils.py | 36 +++++++++++++++++++++++---- tests/test_observe_describe_schema.py | 28 +++++++++++++++++++++ tests/test_observe_utils.py | 24 ++++++++++++++++++ 5 files changed, 124 insertions(+), 7 deletions(-) diff --git a/docs/OBSERVE.md b/docs/OBSERVE.md index cf33ef0c9..1634ab8c4 100644 --- a/docs/OBSERVE.md +++ b/docs/OBSERVE.md @@ -64,9 +64,27 @@ What remains in this package is the home-side ingest-and-processing pipeline: - **sense.py** — File watcher that dispatches transcription and description jobs - **transcribe/** — Audio transcription with sentence-level embeddings -- **describe.py** — Vision analysis with Gemini, category-based prompts +- **describe.py** — Provider-routed vision analysis with category-based prompts - **categories/** — Category-specific prompts for screen content (see [SCREEN_CATEGORIES.md](SCREEN_CATEGORIES.md)) +### Vision input sizing + +Image sizing is phase- and runtime-specific. The application never enlarges an +input image. + +| Path | Bundled Linux Qwen sizing | Other providers/platforms | +|---|---|---| +| Frame categorization (`observe.describe.frame`) | 1024 image-token area ceiling, with the standing 1920px longest-side ceiling | standing 1920px ceiling | +| Category extraction (`observe.describe.`) | standing 1920px ceiling | standing 1920px ceiling | +| Still depiction (`observe.depict`) | standing 1920px ceiling | standing 1920px ceiling | +| Image/document import vision | model preprocessor defaults | model preprocessor defaults | + +The 1024 categorization ceiling is intentionally limited to the bundled Linux +Qwen/llama.cpp path. Apple MLX and configured BYO OpenAI-compatible endpoints +retain their existing preprocessing. Detailed extraction also retains current +sizing: the 2026-07-11 frozen fidelity gate found that 1024 reduced fine-text +fact recall even though it passed categorization. + ## Standalone Observers Each observer is a standalone package in its own repo (see the Observer Architecture table above), with its own capture internals and lifecycle: diff --git a/solstone/observe/describe.py b/solstone/observe/describe.py index fcccd462f..d4ca9a0fc 100644 --- a/solstone/observe/describe.py +++ b/solstone/observe/describe.py @@ -72,6 +72,22 @@ SCENE_CUT_THRESHOLD = 25 # Minimum wall-clock gap (seconds) between kept frames for non-scene-cut, # dHash-qualified frames; closer arrivals are stride-dropped. MIN_STRIDE_SECONDS = 5.0 +# Frozen Fedora gate (2026-07-11): 1024 is the lowest conservative image-token +# ceiling for bundled Linux Qwen categorization. Detail extraction remains at +# current sizing because 1024 lost fine-text fidelity. +LOCAL_QWEN_CATEGORIZATION_IMAGE_TOKENS = 1024 + + +def _categorization_image_token_budget(provider: str) -> int | None: + """Return the proven image cap only for the bundled Linux Qwen path.""" + if provider != "local" or not sys.platform.startswith("linux"): + return None + + from solstone.think.providers.local_endpoint import resolve_local_endpoint + + if not resolve_local_endpoint().is_bundled: + return None + return LOCAL_QWEN_CATEGORIZATION_IMAGE_TOKENS def _winnow_decision( @@ -753,12 +769,17 @@ class VideoProcessor: if frame_provider == NO_BRAIN_PROVIDER: logger.info("No thinking engine selected; deferring frame description") return + categorization_image_tokens = _categorization_image_token_budget( + frame_provider + ) # Create vision requests for all qualified frames for frame_data in qualified_frames: # Load frame image from bytes - keep it open until request completes frame_img = Image.open(io.BytesIO(frame_data["frame_bytes"])) - frame_img = resize_for_vlm(frame_img) + frame_img = resize_for_vlm( + frame_img, max_image_tokens=categorization_image_tokens + ) req = batch.create( contents=self._user_contents( diff --git a/solstone/observe/utils.py b/solstone/observe/utils.py index 279ccba89..aea3c115f 100644 --- a/solstone/observe/utils.py +++ b/solstone/observe/utils.py @@ -9,6 +9,7 @@ import datetime import hashlib import json import logging +import math import multiprocessing import os import random @@ -41,18 +42,43 @@ PDF_EXTENSIONS = tuple(_PDF_EXTENSIONS) # Pre-resize images to this max longest-side before VLM analysis. Images already # at or below this dimension pass through unchanged. _MAX_VLM_DIM = 1920 +# Qwen's llama.cpp vision path emits one image token per 32x32 pixel region. +_QWEN_IMAGE_TOKEN_EDGE_PX = 32 class AudioDecodeError(RuntimeError): """Audio decode failed in the isolated PyAV worker.""" -def resize_for_vlm(img: "Image.Image") -> "Image.Image": - if max(img.size) <= _MAX_VLM_DIM: +def resize_for_vlm( + img: "Image.Image", *, max_image_tokens: int | None = None +) -> "Image.Image": + """Downsize a VLM input without ever enlarging it. + + ``max_image_tokens`` is the Qwen/llama.cpp image-token area ceiling. The + server rounds dimensions to its patch grid, so observed counts can differ + slightly; this client-side cap remains conservative and phase-specific. + """ + if max_image_tokens is None: + if max(img.size) <= _MAX_VLM_DIM: + return img + resized = img.copy() + resized.thumbnail((_MAX_VLM_DIM, _MAX_VLM_DIM)) + return resized + + from PIL import Image + + width, height = img.size + scale = min(1.0, _MAX_VLM_DIM / max(width, height)) + max_pixels = max_image_tokens * _QWEN_IMAGE_TOKEN_EDGE_PX**2 + scale = min(scale, math.sqrt(max_pixels / (width * height))) + if scale >= 1.0: return img - resized = img.copy() - resized.thumbnail((_MAX_VLM_DIM, _MAX_VLM_DIM)) - return resized + target = ( + max(1, math.floor(width * scale)), + max(1, math.floor(height * scale)), + ) + return img.resize(target, Image.Resampling.LANCZOS) def audio_to_flac_bytes(audio: np.ndarray, sample_rate: int) -> bytes: diff --git a/tests/test_observe_describe_schema.py b/tests/test_observe_describe_schema.py index 5cfb14af3..dc1d494c9 100644 --- a/tests/test_observe_describe_schema.py +++ b/tests/test_observe_describe_schema.py @@ -2,6 +2,7 @@ # Copyright (c) 2026 sol pbc from pathlib import Path +from types import SimpleNamespace from unittest.mock import AsyncMock, patch import pytest @@ -13,6 +14,33 @@ from solstone.think.batch import Batch _SCHEMA = describe_mod._SCHEMA +def test_categorization_image_budget_is_linux_bundled_local_only(monkeypatch): + from solstone.think.providers import local_endpoint + + monkeypatch.setattr(describe_mod.sys, "platform", "linux") + monkeypatch.setattr( + local_endpoint, + "resolve_local_endpoint", + lambda: SimpleNamespace(is_bundled=True), + ) + + assert ( + describe_mod._categorization_image_token_budget("local") + == describe_mod.LOCAL_QWEN_CATEGORIZATION_IMAGE_TOKENS + ) + assert describe_mod._categorization_image_token_budget("google") is None + + monkeypatch.setattr( + local_endpoint, + "resolve_local_endpoint", + lambda: SimpleNamespace(is_bundled=False), + ) + assert describe_mod._categorization_image_token_budget("local") is None + + monkeypatch.setattr(describe_mod.sys, "platform", "darwin") + assert describe_mod._categorization_image_token_budget("local") is None + + def test_describe_schema_file_is_valid_draft_2020_12(): Draft202012Validator.check_schema(_SCHEMA) diff --git a/tests/test_observe_utils.py b/tests/test_observe_utils.py index 00a77098b..dcd6beedd 100644 --- a/tests/test_observe_utils.py +++ b/tests/test_observe_utils.py @@ -3,12 +3,36 @@ """Tests for observe/utils.py functions.""" +from PIL import Image + from solstone.observe.utils import ( assign_monitor_positions, parse_screen_filename, + resize_for_vlm, ) +class TestResizeForVlm: + def test_default_caps_longest_side_without_upscaling(self): + large = Image.new("RGB", (3440, 1440)) + small = Image.new("RGB", (800, 600)) + + resized = resize_for_vlm(large) + + assert resized.size == (1920, 804) + assert resize_for_vlm(small) is small + + def test_image_token_cap_is_area_based_and_never_enlarges(self): + large = Image.new("RGB", (1920, 1080)) + small = Image.new("RGB", (640, 360)) + + resized = resize_for_vlm(large, max_image_tokens=1024) + + assert resized.size == (1365, 768) + assert resized.width * resized.height <= 1024 * 32 * 32 + assert resize_for_vlm(small, max_image_tokens=1024) is small + + class TestAssignMonitorPositions: """Test monitor position assignment algorithm.""" -- 2.51.2