diff --git a/apps/settings/routes.py b/apps/settings/routes.py index 810ffd2d8..62c195024 100644 --- a/apps/settings/routes.py +++ b/apps/settings/routes.py @@ -666,6 +666,7 @@ def get_vision() -> Any: Returns: - max_extractions: Current max extractions setting (default: 20) + - redact: List of redaction rules (default: []) - categories: Dict of category overrides from config - category_defaults: Discovered categories with their defaults """ @@ -691,6 +692,7 @@ def get_vision() -> Any: "max_extractions": describe_config.get( "max_extractions", DEFAULT_MAX_EXTRACTIONS ), + "redact": describe_config.get("redact", []), "categories": describe_config.get("categories", {}), "category_defaults": category_defaults, } @@ -705,6 +707,7 @@ def update_vision() -> Any: Accepts JSON with optional keys: - max_extractions: int (5-100) - Maximum frames to extract + - redact: list[str] - Redaction rules (max 50 rules, 200 chars each) - categories: {name: {importance?, extraction?} | null} - Category overrides Setting a category to null removes its overrides. @@ -747,6 +750,35 @@ def update_vision() -> Any: changed_fields["max_extractions"] = {"old": old_val, "new": max_ext} config["describe"]["max_extractions"] = max_ext + # Handle redact rules update + if "redact" in request_data: + redact = request_data["redact"] + if not isinstance(redact, list) or not all( + isinstance(r, str) for r in redact + ): + return ( + jsonify({"error": "redact must be a list of strings"}), + 400, + ) + if len(redact) > 50: + return ( + jsonify({"error": "redact may contain at most 50 rules"}), + 400, + ) + if any(len(r) > 200 for r in redact): + return ( + jsonify( + {"error": "each redact rule must be 200 characters or fewer"} + ), + 400, + ) + # Filter out empty strings + redact = [r for r in redact if r.strip()] + old_val = old_describe.get("redact") + if old_val != redact: + changed_fields["redact"] = {"old": old_val, "new": redact} + config["describe"]["redact"] = redact + # Handle category overrides if "categories" in request_data: categories_data = request_data["categories"] diff --git a/observe/describe.py b/observe/describe.py index 135bc5749..a02c88102 100644 --- a/observe/describe.py +++ b/observe/describe.py @@ -138,6 +138,30 @@ def _build_categorization_prompt() -> str: ).text +def _build_redact_instruction(rules: List[str]) -> str: + """Build a redaction instruction block from user-configured rules. + + Parameters + ---------- + rules : List[str] + Redaction rules from config, one directive per entry. + + Returns + ------- + str + Formatted instruction block to append to system prompts, + or empty string if no rules. + """ + if not rules: + return "" + + items = "\n".join(f"- {rule}" for rule in rules) + return ( + "\n\nRedaction rules (apply these exactly as written, do not generalize):\n" + + items + ) + + # Discover categories at module level CATEGORIES = _discover_categories() @@ -377,14 +401,18 @@ class VideoProcessor: from think.batch import Batch from think.models import resolve_provider - # Load config for max_extractions + # Load config for max_extractions and redaction rules config = get_config() - max_extractions = config.get("describe", {}).get( + describe_config = config.get("describe", {}) + max_extractions = describe_config.get( "max_extractions", DEFAULT_MAX_EXTRACTIONS ) + redact_instruction = _build_redact_instruction( + describe_config.get("redact", []) + ) # Use dynamically built categorization prompt - system_instruction = CATEGORIZATION_PROMPT + system_instruction = CATEGORIZATION_PROMPT + redact_instruction # Process video to get qualified frames (synchronous) qualified_frames = self.process() @@ -699,7 +727,7 @@ class VideoProcessor: entities=True, ), model=cat_model, - system_instruction=cat_meta["prompt"], + system_instruction=cat_meta["prompt"] + redact_instruction, json_output=is_json, max_output_tokens=10240 if is_json else 8192, thinking_budget=6144 if is_json else 4096, diff --git a/pyproject.toml b/pyproject.toml index c4bdecf75..fcc1d9cfe 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -106,7 +106,7 @@ py-modules = ["sol"] [tool.setuptools.package-data] apps = ["*/templates/*.html", "*/muse/*.md"] -think = ["*.md", "templates/*.md"] +think = ["*.md", "*.json", "templates/*.md"] muse = ["*.md", "*.py"] observe = ["*.md", "categories/*.md", "transcribe/*.md"] convey = [ diff --git a/tests/fixtures/journal/config/journal.json b/tests/fixtures/journal/config/journal.json index b2f7dd208..2d4547179 100644 --- a/tests/fixtures/journal/config/journal.json +++ b/tests/fixtures/journal/config/journal.json @@ -2,6 +2,7 @@ "identity": { "name": "Test User", "preferred": "Tester", + "bio": "", "pronouns": { "subject": "they", "object": "them", @@ -48,6 +49,9 @@ } } }, + "describe": { + "redact": ["replace test-secret-key with ***"] + }, "env": { "TEST_CONFIG_ENV_VAR": "from_config" } diff --git a/tests/test_config.py b/tests/test_config.py index dbcd05b15..a70076ebc 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -62,6 +62,24 @@ def test_get_config_default_structure(tmp_path, monkeypatch): assert config["identity"]["timezone"] == "" assert config["identity"]["bio"] == "" + # Describe defaults + assert "describe" in config + assert isinstance(config["describe"]["redact"], list) + assert len(config["describe"]["redact"]) > 0 + + +def test_get_config_default_is_deep_copy(tmp_path, monkeypatch): + """Test that modifying returned defaults doesn't affect future calls.""" + monkeypatch.setenv("JOURNAL_PATH", str(tmp_path)) + + config1 = get_config() + config1["identity"]["name"] = "Modified" + config1["describe"]["redact"].append("extra rule") + + config2 = get_config() + assert config2["identity"]["name"] == "" + assert "extra rule" not in config2["describe"]["redact"] + def test_get_config_loads_existing(config_journal, monkeypatch): """Test get_config loads existing configuration.""" @@ -83,18 +101,17 @@ def test_get_config_loads_existing(config_journal, monkeypatch): assert config["identity"]["bio"] == "a software engineer and tester" -def test_get_config_fills_missing_fields(tmp_path, monkeypatch): - """Test get_config fills in missing identity fields.""" +def test_get_config_existing_is_master(tmp_path, monkeypatch): + """Test that existing journal.json is returned as-is without merging defaults.""" monkeypatch.setenv("JOURNAL_PATH", str(tmp_path)) - # Create config with partial identity data + # Create config with only a name - no other identity fields, no describe config_dir = tmp_path / "config" config_dir.mkdir() partial_config = { "identity": { "name": "Partial User", - # Missing other fields } } @@ -104,19 +121,11 @@ def test_get_config_fills_missing_fields(tmp_path, monkeypatch): config = get_config() - # Check that missing fields are filled with defaults + # User's value is preserved assert config["identity"]["name"] == "Partial User" - assert config["identity"]["preferred"] == "" - assert config["identity"]["pronouns"] == { - "subject": "", - "object": "", - "possessive": "", - "reflexive": "", - } - assert config["identity"]["aliases"] == [] - assert config["identity"]["email_addresses"] == [] - assert config["identity"]["timezone"] == "" - assert config["identity"]["bio"] == "" + # Missing fields are NOT filled from defaults - journal.json is master + assert "preferred" not in config["identity"] + assert "describe" not in config def test_get_config_uses_default_when_journal_path_empty(monkeypatch, tmp_path): @@ -155,6 +164,7 @@ def test_get_config_handles_invalid_json(tmp_path, monkeypatch): "reflexive": "", } assert config["identity"]["bio"] == "" + assert "describe" in config def test_get_config_with_fixtures(): @@ -164,7 +174,7 @@ def test_get_config_with_fixtures(): config = get_config() - # Should return default structure since fixtures doesn't have config yet + # Fixtures has journal.json - returned as-is assert "identity" in config assert isinstance(config["identity"]["name"], str) assert isinstance(config["identity"]["preferred"], str) diff --git a/tests/test_describe_config.py b/tests/test_describe_config.py index b6139f47c..79ff396c3 100644 --- a/tests/test_describe_config.py +++ b/tests/test_describe_config.py @@ -4,6 +4,7 @@ """Tests for observe/describe.py category discovery and configuration.""" from observe import describe as describe_module +from observe.describe import _build_redact_instruction def test_categories_discovered(): @@ -78,3 +79,39 @@ def test_categorization_prompt_alphabetical(): # Should be sorted assert categories == sorted(categories) + + +def test_redact_instruction_empty(): + """Test that empty/missing redact list returns empty string.""" + assert _build_redact_instruction([]) == "" + assert _build_redact_instruction(None) == "" + + +def test_redact_instruction_format(): + """Test that redact instruction formats rules correctly.""" + rules = [ + "use *** instead of any visible passwords", + "replace personal email addresses with [redacted]", + ] + result = _build_redact_instruction(rules) + + # Should contain the header + assert "Redaction rules" in result + assert "do not generalize" in result + + # Should contain each rule as a bullet + for rule in rules: + assert f"- {rule}" in result + + +def test_redact_instruction_no_vague_language(): + """Test that redact instruction doesn't add vague privacy language.""" + rules = ["replace bank account numbers with ***"] + result = _build_redact_instruction(rules) + + # Should not contain vague directives the model could over-apply + lower = result.lower() + assert "sensitive" not in lower + assert "confidential" not in lower + assert "personally identifiable" not in lower + assert "pii" not in lower diff --git a/think/journal_default.json b/think/journal_default.json new file mode 100644 index 000000000..4e71a9898 --- /dev/null +++ b/think/journal_default.json @@ -0,0 +1,23 @@ +{ + "identity": { + "name": "", + "preferred": "", + "bio": "", + "pronouns": { + "subject": "", + "object": "", + "possessive": "", + "reflexive": "" + }, + "aliases": [], + "email_addresses": [], + "timezone": "" + }, + "describe": { + "redact": [ + "use *** instead of any visible passwords, credentials, keys, tokens, and secrets", + "completely omit and ignore any NSFW or adult content, do not mention or note it", + "use *** instead of any visible credit card numbers, bank account numbers, and government ID numbers" + ] + } +} diff --git a/think/utils.py b/think/utils.py index 05636d732..46a7c82d9 100644 --- a/think/utils.py +++ b/think/utils.py @@ -11,6 +11,7 @@ agent configs, etc.) have been moved to think/muse.py. from __future__ import annotations import argparse +import copy import json import logging import os @@ -363,59 +364,55 @@ def _format_time(t: datetime.time) -> str: return datetime.combine(datetime.today(), t).strftime("%I:%M %p").lstrip("0") +def _load_default_config() -> dict[str, Any]: + """Load the default journal configuration from journal_default.json. + + Returns + ------- + dict + Default configuration structure. + """ + default_path = Path(__file__).parent / "journal_default.json" + with open(default_path, "r", encoding="utf-8") as f: + return json.load(f) + + +# Cached default config (loaded once at first use) +_default_config: dict[str, Any] | None = None + + def get_config() -> dict[str, Any]: """Return the journal configuration from config/journal.json. + When no journal.json exists, returns a deep copy of the defaults from + think/journal_default.json. Once journal.json exists it is the master + and is returned as-is with no merging of defaults. + Returns ------- dict - Journal configuration with at least an 'identity' key containing - name, preferred, bio, pronouns, aliases, email_addresses, and - timezone fields. Returns default empty structure if config file doesn't exist. + Journal configuration. """ - # Default identity structure - defined once - default_identity = { - "name": "", - "preferred": "", - "bio": "", - "pronouns": { - "subject": "", - "object": "", - "possessive": "", - "reflexive": "", - }, - "aliases": [], - "email_addresses": [], - "timezone": "", - } + global _default_config + if _default_config is None: + _default_config = _load_default_config() journal = get_journal() config_path = Path(journal) / "config" / "journal.json" - # Return default structure if file doesn't exist + # Return defaults when no config file exists yet if not config_path.exists(): - return {"identity": default_identity.copy()} + return copy.deepcopy(_default_config) try: with open(config_path, "r", encoding="utf-8") as f: - config = json.load(f) - - # Ensure identity section exists with all required fields - if "identity" not in config: - config["identity"] = {} - - # Fill in any missing fields with defaults - for key, default in default_identity.items(): - if key not in config["identity"]: - config["identity"][key] = default - - return config + return json.load(f) except (json.JSONDecodeError, OSError) as exc: - # Log error but return default structure to avoid breaking callers + # Log error but return defaults to avoid breaking callers logging.getLogger(__name__).warning( "Failed to load config from %s: %s", config_path, exc ) - return {"identity": default_identity.copy()} + return copy.deepcopy(_default_config) def _append_task_log(dir_path: str | Path, message: str) -> None: