From 6af27873a743337ce1575b38d9eee2ebcba2c08d Mon Sep 17 00:00:00 2001 From: Jer Miller Date: Tue, 21 Apr 2026 21:42:16 -0600 Subject: [PATCH] providers+logs: cap Gemini output tokens at 65536, fix sol health logs chronicle path - Lower default `thinking_budget` in `think/talents.py` from `8192 * 3` to `8192 * 2` (16384); new default total (16384 + 49152) equals Gemini's 65536 cap. - Clamp outgoing `max_output_tokens` in `think/providers/google.py::_build_generate_config` to `<= GEMINI_MAX_OUTPUT_TOKENS` (65536) with a WARNING log; unit-tested at boundary and above. - Drop `thinking_budget` / `max_output_tokens` frontmatter overrides from `talent/sense.md` so it inherits the new defaults. - Fix `think/logs_cli.py::get_today_health_dir` to read from `journal/chronicle//health/` (missed during the 173c1773 chronicle rename); updated `tests/test_logs_cli.py::make_journal` fixture to match. Regenerated `tests/baselines/api/stats/stats.json` after the sense.md change. Co-Authored-By: Claude Opus 4.7 (1M context) --- talent/sense.md | 2 -- tests/baselines/api/stats/stats.json | 2 -- tests/test_logs_cli.py | 4 +-- tests/test_providers_google.py | 52 ++++++++++++++++++++++++++++ think/logs_cli.py | 6 ++-- think/providers/google.py | 14 ++++++++ think/talents.py | 2 +- 7 files changed, 71 insertions(+), 11 deletions(-) create mode 100644 tests/test_providers_google.py diff --git a/talent/sense.md b/talent/sense.md index faf047dcd..63a3a69ec 100644 --- a/talent/sense.md +++ b/talent/sense.md @@ -7,8 +7,6 @@ "schedule": "segment", "priority": 5, "tier": 3, - "thinking_budget": 4096, - "max_output_tokens": 4096, "output": "json", "schema": "sense.schema.json", "load": {"transcripts": true, "percepts": true, "talents": false} diff --git a/tests/baselines/api/stats/stats.json b/tests/baselines/api/stats/stats.json index 58ad37417..44dc7e556 100644 --- a/tests/baselines/api/stats/stats.json +++ b/tests/baselines/api/stats/stats.json @@ -244,7 +244,6 @@ "talents": false, "transcripts": true }, - "max_output_tokens": 4096, "mtime": 0, "output": "json", "path": "/talent/sense.md", @@ -252,7 +251,6 @@ "schedule": "segment", "schema": "sense.schema.json", "source": "system", - "thinking_budget": 4096, "tier": 3, "title": "Segment Sense", "type": "generate" diff --git a/tests/test_logs_cli.py b/tests/test_logs_cli.py index ccd439412..0c9e04e39 100644 --- a/tests/test_logs_cli.py +++ b/tests/test_logs_cli.py @@ -10,7 +10,7 @@ import pytest def make_journal(tmp_path, day, services, supervisor_lines=None): """Create a synthetic journal with health logs.""" - health_dir = tmp_path / day / "health" + health_dir = tmp_path / "chronicle" / day / "health" health_dir.mkdir(parents=True) for name, lines in services.items(): @@ -25,7 +25,7 @@ def make_journal(tmp_path, day, services, supervisor_lines=None): for name in services: journal_sym = journal_health / f"{name}.log" - journal_sym.symlink_to(f"../{day}/health/ref_{name}.log") + journal_sym.symlink_to(f"../chronicle/{day}/health/ref_{name}.log") if supervisor_lines is not None: sup = journal_health / "supervisor.log" diff --git a/tests/test_providers_google.py b/tests/test_providers_google.py new file mode 100644 index 000000000..88a377a5d --- /dev/null +++ b/tests/test_providers_google.py @@ -0,0 +1,52 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Copyright (c) 2026 sol pbc + +import logging + + +def test_build_generate_config_passes_through_at_cap(caplog): + from think.providers.google import GEMINI_MAX_OUTPUT_TOKENS, _build_generate_config + + with caplog.at_level(logging.WARNING, logger="think.providers.google"): + config = _build_generate_config( + temperature=0.3, + max_output_tokens=49152, + system_instruction=None, + json_output=False, + thinking_budget=16384, + ) + + warnings = [ + record + for record in caplog.records + if record.name == "think.providers.google" and record.levelno == logging.WARNING + ] + assert config.max_output_tokens == GEMINI_MAX_OUTPUT_TOKENS + assert config.thinking_config.thinking_budget == 16384 + assert warnings == [] + + +def test_build_generate_config_clamps_and_warns_once(caplog): + from think.providers.google import GEMINI_MAX_OUTPUT_TOKENS, _build_generate_config + + with caplog.at_level(logging.WARNING, logger="think.providers.google"): + config = _build_generate_config( + temperature=0.3, + max_output_tokens=49152, + system_instruction=None, + json_output=False, + thinking_budget=24576, + ) + + warnings = [ + record + for record in caplog.records + if record.name == "think.providers.google" and record.levelno == logging.WARNING + ] + assert config.max_output_tokens <= GEMINI_MAX_OUTPUT_TOKENS + assert config.max_output_tokens == GEMINI_MAX_OUTPUT_TOKENS + assert config.thinking_config.thinking_budget == 16384 + assert len(warnings) == 1 + assert "max_output_tokens=49152" in warnings[0].message + assert "thinking_budget=24576" in warnings[0].message + assert "clamped_thinking_budget=16384" in warnings[0].message diff --git a/think/logs_cli.py b/think/logs_cli.py index 984406cf6..78d1d1c71 100644 --- a/think/logs_cli.py +++ b/think/logs_cli.py @@ -23,7 +23,7 @@ from datetime import datetime, timedelta from pathlib import Path from typing import NamedTuple -from think.utils import get_journal, setup_cli +from think.utils import day_path, get_journal, setup_cli _DIM = "\033[2m" _RESET = "\033[0m" @@ -108,9 +108,7 @@ def compile_grep(pattern: str) -> re.Pattern[str]: def get_today_health_dir() -> Path | None: - journal = Path(os.path.expanduser(get_journal())) - today = datetime.now().strftime("%Y%m%d") - health_dir = journal / today / "health" + health_dir = day_path(create=False) / "health" return health_dir if health_dir.is_dir() else None diff --git a/think/providers/google.py b/think/providers/google.py index 9496b516f..b09f3d5ee 100644 --- a/think/providers/google.py +++ b/think/providers/google.py @@ -56,6 +56,7 @@ from .shared import ( safe_raw, ) +GEMINI_MAX_OUTPUT_TOKENS = 65536 _DEFAULT_MAX_TOKENS = 8192 _DEFAULT_MODEL = GEMINI_FLASH @@ -244,6 +245,19 @@ def _build_generate_config( """ # Compute total tokens: output + thinking budget total_tokens = max_output_tokens + (thinking_budget or 0) + if total_tokens > GEMINI_MAX_OUTPUT_TOKENS: + clamped_max_output = min(max_output_tokens, GEMINI_MAX_OUTPUT_TOKENS) + clamped_thinking = max(0, GEMINI_MAX_OUTPUT_TOKENS - clamped_max_output) + logging.getLogger(__name__).warning( + "Clamping Gemini token budget: max_output_tokens=%s thinking_budget=%s " + "clamped_max_output_tokens=%s clamped_thinking_budget=%s", + max_output_tokens, + thinking_budget, + clamped_max_output, + clamped_thinking, + ) + thinking_budget = clamped_thinking + total_tokens = clamped_max_output + clamped_thinking config_args: dict[str, Any] = { "temperature": temperature, diff --git a/think/talents.py b/think/talents.py index 49b66c43a..2a31b732b 100644 --- a/think/talents.py +++ b/think/talents.py @@ -938,7 +938,7 @@ async def _execute_generate( output_format = config.get("output") # Get generation parameters from config (set in frontmatter) - thinking_budget = config.get("thinking_budget") or 8192 * 3 + thinking_budget = config.get("thinking_budget") or 8192 * 2 max_output_tokens = config.get("max_output_tokens") or 8192 * 6 is_json_output = output_format == "json" -- 2.51.2