diff --git a/core/fixtures/cogitate_contract.json b/core/fixtures/cogitate_contract.json index f99673822..7ba3cbff0 100644 --- a/core/fixtures/cogitate_contract.json +++ b/core/fixtures/cogitate_contract.json @@ -52,8 +52,8 @@ ], "runtime_preamble": { "algorithm": "sha256", - "byte_length": 1745, - "digest": "80fb4b8b98ee800489ca92728b0a49ffd5dfdd35f1af7cc129a7831273db0568", + "byte_length": 1989, + "digest": "6614e3fd060f29e7c3cb0ed063e43e370a7efefd2579974080fc8caec65ca591", "encoding": "utf-8" } } diff --git a/docs/COGITATE.md b/docs/COGITATE.md index d65444c45..41abf1084 100644 --- a/docs/COGITATE.md +++ b/docs/COGITATE.md @@ -61,18 +61,22 @@ The journal-root cwd matters operationally because the `sol` tool inherits it, b ## The `sol` CLI is the authoritative talent-to-journal contract A talent reaches journal functionality by emitting `sol` / `sol call ...` -command lines. The runtime parses each tool call as one command-line invocation -and executes that argv directly; it is not an arbitrary shell. The `sol` CLI -handlers are the **single authoritative translation layer** between a talent and -the journal: they turn CLI syntax into convey API calls. A talent therefore: +command lines. The approved host command families `journal identity ...`, +`journal health ...`, and `journal talent ...` also run directly and must not be +prefixed with `sol` or `sol call`. The runtime parses each tool call as one +command-line invocation and executes that argv directly; it is not an arbitrary +shell. The CLI handlers are the **single authoritative translation layer** +between a talent and the journal: they turn CLI syntax into journal operations. +A talent therefore: - **never** talks to the convey HTTP API directly, and - **never** assumes a database, a socket, or any journal access other than the - `sol` CLI and the bounded raw-read tools below. + documented command forms and the bounded raw-read tools below. -The tool call shape is `sol(command="sol call activities list")`. The command -policy (below) constrains what may run and rejects shell composition such as -pipes, redirects, chaining, and command substitution. +The tool call shape is `sol(command="sol call activities list")`, or +`sol(command="journal identity partner")` for an approved host command family. +The command policy (below) constrains what may run and rejects shell composition +such as pipes, redirects, chaining, and command substitution. ## Reads: domain reads vs raw evidence reads @@ -103,16 +107,16 @@ default gate, and a talent never fails for reading an undeclared journal file. > access tier. Check the run's actual tool schema > (`journal talent show --prompt`) for what a specific talent receives. -## Writes: only through `sol` domain commands +## Writes: only through approved journal commands -A talent **writes journal state only through `sol` domain commands** (the +A talent **writes journal state only through approved journal commands** (the `sol call ...` verbs for the domain it is updating, e.g. -`sol call entities update ...`). There is **no general-purpose write tool** and no -raw-write tier — write-class tools are denied by policy. The mechanic is -centralized in the domain command (which routes through the convey API to the -single write-owner for that data); a talent gets the *capability* to update a -domain, never an arbitrary file-write path. Persistence that does not go through a -`sol` domain command does not happen. +`sol call entities update ...`, plus approved direct host commands when a prompt +names one). There is **no general-purpose write tool** and no raw-write tier — +write-class tools are denied by policy. The mechanic is centralized in the owning +command; a talent gets the *capability* to update a domain, never an arbitrary +file-write path. Persistence that does not go through an approved journal command +does not happen. ## Access tiers @@ -126,7 +130,7 @@ enforcement are layered on top of it. | `normal` | default cogitate talents | the `sol` tool (`sol` / `sol call`), the bounded raw-read tier, a finalization tool | | `system-read` | diagnostics boundary for scoped operational evidence | no cogitate talent claims it today (steward was demoted to a deterministic renderer + `lite` generate); the tier remains the declared diagnostics boundary and extension point, with scoped evidence arriving through a talent pre-hook rather than an extra model read tool | | `outbound` | comms-like talents that may submit something that leaves the machine (e.g. `support`) | the `sol` tool (`sol` / `sol call`) and a finalization tool, plus submit-capable support commands gated on per-send owner approval supplied only by a human-initiated chat launch; no raw-read tier — drafts and evidence go through `sol` domain commands | -| `synthesis` | pure sol-surface synthesis talents (e.g. `weekly_reflection`, `partner`) whose source of record is `sol call journal` / `sol call activities`, not the raw journal tree | the `sol` tool (`sol` / `sol call`) and a finalization tool; **no raw-read tier and no submit** — same as `outbound` minus the outbound submit capability. Removing the raw-read tools keeps a synthesis talent from spelunking `chronicle/` / `talents/` / `facets/` and burning its budget instead of reaching the journal through `sol` | +| `synthesis` | pure command-surface synthesis talents (e.g. `weekly_reflection`, `partner`) whose source of record is a documented command form, not the raw journal tree | the `sol` tool (`sol` / `sol call`, plus approved direct `journal` families when a prompt names one) and a finalization tool; **no raw-read tier and no submit** — same as `outbound` minus the outbound submit capability. Removing the raw-read tools keeps a synthesis talent from spelunking `chronicle/` / `talents/` / `facets/` and burning its budget instead of using documented commands | Policy denies support send verbs (`create`, `reply`, `attach`, `feedback`) for `normal` / `system-read` runs. `outbound` runs may use those verbs only when the @@ -157,7 +161,7 @@ Every talent has exactly one finalization mode (`TALENT_FINALIZATION_MODES` in |---|---|---| | `emit_final` | scheduled / output talents (an `output_path` is set, or the schedule is `daily` / `weekly` / `activity`) | the run is accepted **only** from the `emit_final` tool; `FinishTool` is disabled; a finish with no emitted final is an error | | `FinishTool` | manual / no-output talents | the run signals completion through OpenHands' built-in finish tool | -| `quiet` | side-effect-only talents | the talent has already persisted its work through `sol` domain commands and finishes with no output | +| `quiet` | side-effect-only talents | the talent has already persisted its work through approved journal commands and finishes with no output | A manual talent's prompt should state which of these it expects so the behavior is explicit rather than inferred. @@ -167,8 +171,9 @@ explicit rather than inferred. A talent prompt must not assume any of the following — they are named here so prompts can be checked against them: -- **bare `journal ...` commands** (e.g. `journal identity`, `journal health`, - `journal navigate`) — reach the journal through `sol` / `sol call`; +- **other bare `journal ...` families** (e.g. `journal search`, `journal facet`, + `journal navigate`) — reach the journal through the documented `sol` / + `sol call` form; - **raw `cat` / `ls` / arbitrary shell file reads** — use the raw-read tools when present, not shell file access; - **auto-loaded skills, `AGENTS.md`, `CLAUDE.md`, or `GEMINI.md`** — these are @@ -191,10 +196,11 @@ verbatim text: You are a solstone cogitate talent running inside the live system. This runtime contract is authoritative; do not assume capabilities beyond it. - Reach the journal through the `sol` command line: emit `sol` / `sol call ...` command lines, e.g. sol(command="sol call activities list"). The runtime runs each call as a single parsed command-line invocation, not an arbitrary shell. The `sol` CLI is the one authoritative path between you and the journal; never assume direct database, socket, or HTTP access. -- Write journal state only through `sol` domain commands (the `sol call ...` verbs for the data you own). There is no general-purpose write tool; persistence that does not go through a `sol` domain command will not happen. +- The approved host command families are identity, health, talent; run them directly as `journal ...` through the same tool, never prefixed with `sol` or `sol call`. +- Write journal state only through approved journal commands (`sol call ...` verbs for the data you own, plus approved direct host commands when a prompt names one). There is no general-purpose write tool; persistence that does not go through an approved journal command will not happen. - Raw evidence reads use the provided read tools (`read_file`, `list_directory`, `glob`, `grep_search`), bounded to the journal root: a denylist (`.git`, caches, credentials, virtualenvs, `node_modules`) and per-call / per-run caps apply. Prefer `sol call` reads; use raw reads only for evidence that has no `sol` command. - Finalize as your run is configured: call `emit_final` when an `emit_final` tool is present; otherwise finish through the built-in finish tool; a side-effect-only talent that has already persisted its work finishes quietly with no output. -- Do not assume tools or context you were not given: no bare `journal ...` commands, no raw `cat` / `ls` / shell file reads, no shell composition (pipes, redirects, chaining, or command substitution; one command per call), no auto-loaded skills or AGENTS.md / CLAUDE.md, no browser or web access, no MCP tools, and no delegating to sub-agents. Any guidance file is a normal journal file with no special status; this contract is your source of truth. +- Do not assume tools or context you were not given: no other bare `journal ...` family, no raw `cat` / `ls` / shell file reads, no shell composition (pipes, redirects, chaining, or command substitution; one command per call), no auto-loaded skills or AGENTS.md / CLAUDE.md, no browser or web access, no MCP tools, and no delegating to sub-agents. Any guidance file is a normal journal file with no special status; this contract is your source of truth. ``` ## Where this is wired (for maintainers) diff --git a/scripts/check_cogitate_prompts.py b/scripts/check_cogitate_prompts.py index 1ded8f187..0afbbcace 100644 --- a/scripts/check_cogitate_prompts.py +++ b/scripts/check_cogitate_prompts.py @@ -37,13 +37,13 @@ from pathlib import Path import frontmatter import solstone.think.talent as talent +from solstone.think.cogitate_contract import ( + COGITATE_JOURNAL_COMMANDS, + cogitate_journal_command_list, +) ROOT = Path(__file__).resolve().parent.parent -# Must equal solstone/think/cogitate_policy.py:_JOURNAL_COMMANDS -# (cogitate_policy.py:21). Duplicated here intentionally; no shared import. -ALLOWED_JOURNAL_COMMANDS = frozenset({"identity", "health", "talent"}) - # Must equal solstone/think/cogitate_policy.py:_READ_TOOLS # (cogitate_policy.py:24). Duplicated here intentionally; no shared import. READ_TOOLS = frozenset({"read_file", "glob", "list_directory", "grep_search"}) @@ -87,7 +87,9 @@ UNSUPPORTED_FLAGS: list[tuple[tuple[str, ...], str, str]] = [] ALLOWLIST: dict[tuple[str, str], int] = {} JOURNAL_ALTERNATIVE = ( - "use `journal` with one of {identity, health, talent}, or use `sol`/`sol call`" + "use `journal` with one of {" + + cogitate_journal_command_list() + + "}, or use `sol`/`sol call`" ) READ_ALTERNATIVE = ( "use a bounded read tool: read_file, list_directory, glob, or grep_search" @@ -246,7 +248,7 @@ def classify_span(command: str) -> list[tuple[str, str]]: if ( tokens[0] == "journal" and len(tokens) >= 2 - and tokens[1] not in ALLOWED_JOURNAL_COMMANDS + and tokens[1] not in COGITATE_JOURNAL_COMMANDS ): findings.append( ( diff --git a/solstone/talent/partner.md b/solstone/talent/partner.md index eaa8c9f64..c1bd29493 100644 --- a/solstone/talent/partner.md +++ b/solstone/talent/partner.md @@ -20,8 +20,8 @@ This is not a conversation. Gather data, observe patterns, update the profile, t ## Step 1: Read current state -Read the current profile with `journal identity partner` — the settled -`sol`-surface read form for `identity/partner.md`. +Read the current profile with `journal identity partner` through the provided +`sol` tool — it is the approved direct host read command for `identity/partner.md`. Note which sections have real observations vs `[observing]` placeholders. @@ -40,8 +40,8 @@ and query each source. If a source returns empty or errors, skip it — gaps are For each of the five profile sections, analyze the gathered data and write observations if you have sufficient evidence. Use `journal identity partner --update-section` -for each section you update — it is the owned write command for `partner.md` -(there is no `sol call` verb for it yet). +through the provided `sol` tool for each section you update — it is the approved +direct host write command for `partner.md`. ### Section guidance diff --git a/solstone/think/cogitate_contract.py b/solstone/think/cogitate_contract.py index 914ca60bc..99c9c5bc2 100644 --- a/solstone/think/cogitate_contract.py +++ b/solstone/think/cogitate_contract.py @@ -15,14 +15,23 @@ from __future__ import annotations from dataclasses import dataclass from typing import Any -COGITATE_RUNTIME_PREAMBLE = """\ +COGITATE_JOURNAL_COMMANDS = ("identity", "health", "talent") + + +def cogitate_journal_command_list() -> str: + """Return the approved direct journal command families for prompt copy.""" + return ", ".join(COGITATE_JOURNAL_COMMANDS) + + +COGITATE_RUNTIME_PREAMBLE = f"""\ You are a solstone cogitate talent running inside the live system. This runtime contract is authoritative; do not assume capabilities beyond it. - Reach the journal through the `sol` command line: emit `sol` / `sol call ...` command lines, e.g. sol(command="sol call activities list"). The runtime runs each call as a single parsed command-line invocation, not an arbitrary shell. The `sol` CLI is the one authoritative path between you and the journal; never assume direct database, socket, or HTTP access. -- Write journal state only through `sol` domain commands (the `sol call ...` verbs for the data you own). There is no general-purpose write tool; persistence that does not go through a `sol` domain command will not happen. +- The approved host command families are {cogitate_journal_command_list()}; run them directly as `journal ...` through the same tool, never prefixed with `sol` or `sol call`. +- Write journal state only through approved journal commands (`sol call ...` verbs for the data you own, plus approved direct host commands when a prompt names one). There is no general-purpose write tool; persistence that does not go through an approved journal command will not happen. - Raw evidence reads use the provided read tools (`read_file`, `list_directory`, `glob`, `grep_search`), bounded to the journal root: a denylist (`.git`, caches, credentials, virtualenvs, `node_modules`) and per-call / per-run caps apply. Prefer `sol call` reads; use raw reads only for evidence that has no `sol` command. - Finalize as your run is configured: call `emit_final` when an `emit_final` tool is present; otherwise finish through the built-in finish tool; a side-effect-only talent that has already persisted its work finishes quietly with no output. -- Do not assume tools or context you were not given: no bare `journal ...` commands, no raw `cat` / `ls` / shell file reads, no shell composition (pipes, redirects, chaining, or command substitution; one command per call), no auto-loaded skills or AGENTS.md / CLAUDE.md, no browser or web access, no MCP tools, and no delegating to sub-agents. Any guidance file is a normal journal file with no special status; this contract is your source of truth. +- Do not assume tools or context you were not given: no other bare `journal ...` family, no raw `cat` / `ls` / shell file reads, no shell composition (pipes, redirects, chaining, or command substitution; one command per call), no auto-loaded skills or AGENTS.md / CLAUDE.md, no browser or web access, no MCP tools, and no delegating to sub-agents. Any guidance file is a normal journal file with no special status; this contract is your source of truth. """ COGITATE_DIAGNOSTIC_PREAMBLE = """\ @@ -71,10 +80,11 @@ _ACCESS_TIER_CAPABILITIES: dict[str, AccessCapabilities] = { "normal": AccessCapabilities(sol=True, reads=True, submit=False), "system-read": AccessCapabilities(sol=True, reads=True, submit=False), "outbound": AccessCapabilities(sol=True, reads=False, submit=True), - # Pure sol-surface synthesis: the journal is reached only through `sol` - # domain commands, with no raw-filesystem read tier and no outbound submit. + # Pure command-surface synthesis: the journal is reached only through + # approved journal commands, with no raw-filesystem read tier and no + # outbound submit. # For synthesis talents (weekly_reflection, partner) whose source of record - # is `sol call journal` / `sol call activities`, not the raw journal tree. + # is a documented command form, not the raw journal tree. "synthesis": AccessCapabilities(sol=True, reads=False, submit=False), "diagnostic": AccessCapabilities(sol=False, reads=False, submit=False), } @@ -109,6 +119,7 @@ def expects_emit_final(config: dict[str, Any]) -> bool: __all__ = [ "AccessCapabilities", "COGITATE_DIAGNOSTIC_PREAMBLE", + "COGITATE_JOURNAL_COMMANDS", "COGITATE_RUNTIME_PREAMBLE", "COGITATE_ACCESS_TIERS", "COGITATE_READ_TOOL_NAMES", @@ -116,5 +127,6 @@ __all__ = [ "TALENT_ACCESS_TIERS", "TALENT_FINALIZATION_MODES", "capabilities_for_access_tier", + "cogitate_journal_command_list", "expects_emit_final", ] diff --git a/solstone/think/cogitate_policy.py b/solstone/think/cogitate_policy.py index e9d3b6c88..767ad5faf 100644 --- a/solstone/think/cogitate_policy.py +++ b/solstone/think/cogitate_policy.py @@ -12,6 +12,7 @@ from typing import Any from solstone.think.cogitate_contract import ( COGITATE_ACCESS_TIERS, + COGITATE_JOURNAL_COMMANDS, COGITATE_READ_TOOL_NAMES, capabilities_for_access_tier, ) @@ -77,7 +78,6 @@ def failure_capped(reason_code: str | None, count: int) -> bool: return cap is not None and count >= cap -_JOURNAL_COMMANDS = {"identity", "health", "talent"} _SHELL_OPERATOR_CHARS = frozenset("();<>|&") _WRITE_TOOLS = {"write_file", "replace"} _READ_TOOLS = frozenset(COGITATE_READ_TOOL_NAMES) @@ -95,6 +95,17 @@ EMPTY_COMMAND_DENY = "policy_deny: empty command" RESTRICTED_COMMAND_DENY = ( "policy_deny: run_shell_command restricted to sol or approved journal invocations" ) +HYBRID_JOURNAL_COMMAND_DENY = ( + "policy_deny: `sol call journal {family}` is not a `sol call` verb; " + "`{family}` is an approved host command family — run it directly as `{repair}`" +) +BARE_JOURNAL_REPAIR_DENY = ( + "policy_deny: `journal {family}` is not a host command; run `{repair}` instead" +) +_BARE_JOURNAL_REPAIR_PREFIXES = { + "search": ("sol", "call", "journal", "search"), + "facet": ("sol", "call", "journal", "facet"), +} @dataclass(frozen=True) @@ -136,6 +147,18 @@ def _shell_syntax_violation(command: str) -> bool: return quote is not None +def _hybrid_journal_deny(argv: list[str]) -> str: + family = argv[3] + repair = shlex.join(["journal", *argv[3:]]) + return HYBRID_JOURNAL_COMMAND_DENY.format(family=family, repair=repair) + + +def _bare_journal_repair_deny(argv: list[str]) -> str: + family = argv[1] + repair = shlex.join([*_BARE_JOURNAL_REPAIR_PREFIXES[family], *argv[2:]]) + return BARE_JOURNAL_REPAIR_DENY.format(family=family, repair=repair) + + class MaxTurnsExhausted(RuntimeError): """Raised when the SDK tool loop exceeds its turn ceiling.""" @@ -182,10 +205,26 @@ class CogitatePolicy: if not argv: return CommandDecision(False, EMPTY_COMMAND_DENY, None) + if ( + argv[0:3] == ["sol", "call", "journal"] + and len(argv) >= 4 + and argv[3] in COGITATE_JOURNAL_COMMANDS + ): + return CommandDecision(False, _hybrid_journal_deny(argv), None) + + if ( + argv[0] == "journal" + and len(argv) >= 2 + and argv[1] in _BARE_JOURNAL_REPAIR_PREFIXES + ): + return CommandDecision(False, _bare_journal_repair_deny(argv), None) + if not ( argv[0] == "sol" or ( - argv[0] == "journal" and len(argv) >= 2 and argv[1] in _JOURNAL_COMMANDS + argv[0] == "journal" + and len(argv) >= 2 + and argv[1] in COGITATE_JOURNAL_COMMANDS ) ): return CommandDecision(False, RESTRICTED_COMMAND_DENY, None) diff --git a/solstone/think/providers/cli.py b/solstone/think/providers/cli.py index 456bf3eed..ef4217c60 100644 --- a/solstone/think/providers/cli.py +++ b/solstone/think/providers/cli.py @@ -24,6 +24,7 @@ from solstone.think.cogitate_contract import ( COGITATE_DIAGNOSTIC_PREAMBLE, COGITATE_READ_TOOL_NAMES, COGITATE_RUNTIME_PREAMBLE, + cogitate_journal_command_list, ) from solstone.think.providers.shared import JSONEventCallback, safe_raw from solstone.think.utils import get_project_root, now_ms @@ -88,10 +89,16 @@ async def _drain_line(stream: asyncio.StreamReader) -> None: def cogitate_sol_tool_hint(tool_name: str) -> str: """Return the shell-tool hint for non-write cogitate runs.""" return ( - "When the instructions tell you to run `sol ...` commands, invoke them " - f"through the `{tool_name}` tool. Example: " - f'`{tool_name}(command="sol call activities list")`. ' - "Do not invent or call a tool literally named `sol`." + "When the instructions tell you to run `sol ...` or approved " + f"`journal ...` commands, invoke them through the `{tool_name}` tool. " + "Normal journal access uses `sol` / `sol call ...`; the approved " + "direct `journal` families are " + + cogitate_journal_command_list() + + " and must be run unprefixed as `journal ...`. Examples: " + f'`{tool_name}(command="sol call activities list")`, ' + f'`{tool_name}(command="journal identity partner")`. ' + "Do not invent or call a tool literally named `sol`, and do not " + "rewrite approved `journal` commands as `sol call journal ...`." ) diff --git a/solstone/think/providers/openhands.py b/solstone/think/providers/openhands.py index e3ef0d04f..6898d66ff 100644 --- a/solstone/think/providers/openhands.py +++ b/solstone/think/providers/openhands.py @@ -32,6 +32,7 @@ from typing import Any from solstone.log_policy import apply_http_logging_policy, snapshot_root_logging from solstone.think.cogitate_contract import ( capabilities_for_access_tier, + cogitate_journal_command_list, expects_emit_final, ) from solstone.think.cogitate_policy import ( @@ -779,8 +780,12 @@ def _ensure_sol_types() -> dict[str, Any]: class SolAction(Action): command: str = Field( description=( - "Single `sol` or approved `journal` command-line invocation to " - "run directly, without a shell." + "Single command-line invocation to run directly, without a shell: " + "use `sol`/`sol call ...` for normal journal access; run approved " + "`journal` families (" + + cogitate_journal_command_list() + + ") directly as `journal ...`, never prefixed with " + "`sol` or `sol call`." ) ) @@ -937,8 +942,12 @@ def _build_sol_tools( ) tool = sol_tool_cls( description=( - "Run one policy-approved `sol` or `journal` command-line invocation " - "directly, without a shell." + "Run one policy-approved command directly, without a shell: use " + "`sol`/`sol call ...` for normal journal access; run approved " + "`journal` families (" + + cogitate_journal_command_list() + + ") directly as `journal ...`, never prefixed with " + "`sol` or `sol call`." ), action_type=sol_action, observation_type=sol_observation, diff --git a/solstone/think/talent_cli.py b/solstone/think/talent_cli.py index 342529544..3c1a3f30a 100644 --- a/solstone/think/talent_cli.py +++ b/solstone/think/talent_cli.py @@ -40,6 +40,7 @@ import frontmatter from solstone.think.cogitate_contract import ( COGITATE_ACCESS_TIERS, + COGITATE_JOURNAL_COMMANDS, COGITATE_READ_TOOL_NAMES, capabilities_for_access_tier, expects_emit_final, @@ -471,9 +472,10 @@ def _discover_cogitate_keys() -> list[str]: def _scan_command_examples(body: str, *, cap: int = 6) -> list[str]: """Scan prompt body text for command examples.""" + journal_alternation = "|".join(COGITATE_JOURNAL_COMMANDS) pattern = re.compile( r"`(?P(?:sol\s+call\s+[^\n`]+|journal\s+" - r"(?:identity|health|talent)\b[^\n`]*))`" + rf"(?:{journal_alternation})\b[^\n`]*))`" ) seen: set[str] = set() result: list[str] = [] diff --git a/tests/test_check_cogitate_prompts.py b/tests/test_check_cogitate_prompts.py index 506ef4577..6cde83f76 100644 --- a/tests/test_check_cogitate_prompts.py +++ b/tests/test_check_cogitate_prompts.py @@ -79,8 +79,7 @@ def test_bare_journal_flags_fenced_commands() -> None: ( 2, "bare-journal", - "forbidden `journal supervisor`; use `journal` with one of " - "{identity, health, talent}, or use `sol`/`sol call`", + f"forbidden `journal supervisor`; {ccp.JOURNAL_ALTERNATIVE}", ) ] diff --git a/tests/test_cli_provider.py b/tests/test_cli_provider.py index ae7912b1c..d8222f076 100644 --- a/tests/test_cli_provider.py +++ b/tests/test_cli_provider.py @@ -75,8 +75,10 @@ class TestAssemblePrompt: for tool_name in ("Bash", "run_shell_command", "bash"): hint = cogitate_sol_tool_hint(tool_name) assert tool_name in hint - assert "Do not invent or call a tool literally named `sol`." in hint + assert "Do not invent or call a tool literally named `sol`" in hint assert 'command="sol call activities list"' in hint + assert 'command="journal identity partner"' in hint + assert "must be run unprefixed as `journal ...`" in hint def test_assemble_prompt_appends_sol_tool_hint_when_provided(self): body, system = assemble_prompt( diff --git a/tests/test_cogitate_contract.py b/tests/test_cogitate_contract.py index 555a6e288..e25f2285c 100644 --- a/tests/test_cogitate_contract.py +++ b/tests/test_cogitate_contract.py @@ -1,6 +1,7 @@ # SPDX-License-Identifier: AGPL-3.0-only # Copyright (c) 2026 sol pbc +import ast from pathlib import Path import pytest @@ -9,6 +10,7 @@ from solstone.think import cogitate_contract from solstone.think.cogitate_contract import ( COGITATE_ACCESS_TIERS, COGITATE_DIAGNOSTIC_PREAMBLE, + COGITATE_JOURNAL_COMMANDS, COGITATE_READ_TOOL_NAMES, COGITATE_RUNTIME_PREAMBLE, FUTURE_ACCESS_TIERS, @@ -81,6 +83,10 @@ def test_cogitate_vocabulary_lock(): "glob", "grep_search", ) + assert COGITATE_JOURNAL_COMMANDS == ("identity", "health", "talent") + assert cogitate_contract.cogitate_journal_command_list() == ( + "identity, health, talent" + ) assert FUTURE_ACCESS_TIERS == ("code-agent",) assert TALENT_ACCESS_TIERS == ( "normal", @@ -146,6 +152,11 @@ def test_capabilities_for_access_tier_unknown_names_tier(tier): def test_cogitate_runtime_preamble_content_guard(): assert "sol call ..." in COGITATE_RUNTIME_PREAMBLE assert "single parsed command-line invocation" in COGITATE_RUNTIME_PREAMBLE + assert "approved host command families" in COGITATE_RUNTIME_PREAMBLE + assert "journal ..." in COGITATE_RUNTIME_PREAMBLE + assert "never prefixed with `sol` or `sol call`" in COGITATE_RUNTIME_PREAMBLE + assert "no bare `journal ...` commands" not in COGITATE_RUNTIME_PREAMBLE + assert "no other bare `journal ...` family" in COGITATE_RUNTIME_PREAMBLE assert "journal root" in COGITATE_RUNTIME_PREAMBLE assert "node_modules" in COGITATE_RUNTIME_PREAMBLE assert "emit_final" in COGITATE_RUNTIME_PREAMBLE @@ -167,9 +178,11 @@ def test_diagnostic_preamble_omits_journal_tooling(): assert "grep_search" not in system assert "through the `sol` tool" not in system assert "Limit filesystem reads" not in system - assert "through a `sol` domain command" in COGITATE_RUNTIME_PREAMBLE + assert "through an approved journal command" in COGITATE_RUNTIME_PREAMBLE assert "no MCP tools" in COGITATE_RUNTIME_PREAMBLE - assert "no bare `journal ...` commands" in COGITATE_RUNTIME_PREAMBLE + assert "approved host command families" in COGITATE_RUNTIME_PREAMBLE + assert "no bare `journal ...` commands" not in COGITATE_RUNTIME_PREAMBLE + assert "no other bare `journal ...` family" in COGITATE_RUNTIME_PREAMBLE assert "no shell composition" in COGITATE_RUNTIME_PREAMBLE assert "read_file" in COGITATE_RUNTIME_PREAMBLE assert "list_directory" in COGITATE_RUNTIME_PREAMBLE @@ -192,3 +205,44 @@ def test_cogitate_doc_preamble_block_matches_source_constant(): assert closing_fence assert block == COGITATE_RUNTIME_PREAMBLE.rstrip("\n") + + +def test_cogitate_journal_command_vocabulary_not_retyped(): + repo_root = Path(__file__).resolve().parents[1] + allowed = { + Path("solstone/think/cogitate_contract.py"), + Path("tests/test_cogitate_contract.py"), + } + roots = [repo_root / "solstone", repo_root / "scripts", repo_root / "tests"] + findings: list[str] = [] + family_set = set(COGITATE_JOURNAL_COMMANDS) + list_markers = ( + "identity, health, talent", + "{identity, health, talent}", + "identity|health|talent", + ) + + for root in roots: + for path in sorted(root.rglob("*.py")): + rel = path.relative_to(repo_root) + if rel in allowed: + continue + text = path.read_text(encoding="utf-8") + tree = ast.parse(text, filename=str(rel)) + for node in ast.walk(tree): + if isinstance(node, ast.Constant) and isinstance(node.value, str): + if any(marker in node.value for marker in list_markers): + findings.append(f"{rel}:{node.lineno}") + elif isinstance(node, (ast.Tuple, ast.List, ast.Set)): + values = { + elt.value + for elt in node.elts + if isinstance(elt, ast.Constant) and isinstance(elt.value, str) + } + if family_set <= values: + findings.append(f"{rel}:{node.lineno}") + + assert findings == [], ( + "render from `COGITATE_JOURNAL_COMMANDS` in " + "solstone/think/cogitate_contract.py: " + ", ".join(findings) + ) diff --git a/tests/test_cogitate_policy.py b/tests/test_cogitate_policy.py index f0286b877..d70e5f4da 100644 --- a/tests/test_cogitate_policy.py +++ b/tests/test_cogitate_policy.py @@ -73,6 +73,7 @@ def test_policy_denies_write_tools(tmp_path): @pytest.mark.parametrize( "command", [ + "journal identity partner", "journal health logs --since 1h", "journal talent logs --daily -c 10", ], @@ -86,6 +87,84 @@ def test_policy_allows_approved_journal_invocations(tmp_path, command): assert reason == "ok" +@pytest.mark.parametrize( + ("command", "reason"), + [ + ( + "sol call journal identity partner", + "policy_deny: `sol call journal identity` is not a `sol call` verb; " + "`identity` is an approved host command family — run it directly as " + "`journal identity partner`", + ), + ( + "sol call journal identity partner --update-section 'work patterns' --value x", + "policy_deny: `sol call journal identity` is not a `sol call` verb; " + "`identity` is an approved host command family — run it directly as " + "`journal identity partner --update-section 'work patterns' --value x`", + ), + ( + "sol call journal health", + "policy_deny: `sol call journal health` is not a `sol call` verb; " + "`health` is an approved host command family — run it directly as " + "`journal health`", + ), + ( + "sol call journal talent logs", + "policy_deny: `sol call journal talent` is not a `sol call` verb; " + "`talent` is an approved host command family — run it directly as " + "`journal talent logs`", + ), + ], +) +def test_policy_denies_hybrid_journal_invocations_with_repair( + tmp_path, command, reason +): + policy = _policy(tmp_path) + + allowed, actual_reason = policy.check("run_shell_command", {"command": command}) + + assert allowed is False + assert actual_reason == reason + + +@pytest.mark.parametrize( + ("command", "reason"), + [ + ( + 'journal search "partner"', + "policy_deny: `journal search` is not a host command; run " + "`sol call journal search partner` instead", + ), + ( + "journal facet show work", + "policy_deny: `journal facet` is not a host command; run " + "`sol call journal facet show work` instead", + ), + ], +) +def test_policy_denies_bare_journal_search_and_facet_with_repair( + tmp_path, command, reason +): + policy = _policy(tmp_path) + + allowed, actual_reason = policy.check("run_shell_command", {"command": command}) + + assert allowed is False + assert actual_reason == reason + + +def test_policy_allows_sol_call_journal_search_with_hybrid_text(tmp_path): + policy = _policy(tmp_path) + + allowed, reason = policy.check( + "run_shell_command", + {"command": 'sol call journal search "sol call journal identity"'}, + ) + + assert allowed is True + assert reason == "ok" + + @pytest.mark.parametrize( "command", [ diff --git a/tests/test_openhands_sol_tool.py b/tests/test_openhands_sol_tool.py index 9808951f9..4c2f99ed4 100644 --- a/tests/test_openhands_sol_tool.py +++ b/tests/test_openhands_sol_tool.py @@ -7,6 +7,7 @@ import time import pytest +from solstone.think.cogitate_contract import cogitate_journal_command_list from solstone.think.cogitate_policy import CogitatePolicy from solstone.think.providers import local_admission, openhands from solstone.think.providers.local_admission import LocalSlotLease @@ -57,6 +58,27 @@ def _isolated_lease(monkeypatch, tmp_path, *, timeout_s: float = 1.0): ) +def test_sol_tool_descriptions_name_direct_journal_rule( + fake_openhands, + tmp_path, +): + tool, _executor = _sol_tool_and_executor(tmp_path=tmp_path, events=[]) + family_list = cogitate_journal_command_list() + + assert tool.description == ( + "Run one policy-approved command directly, without a shell: use " + "`sol`/`sol call ...` for normal journal access; run approved " + f"`journal` families ({family_list}) directly as " + "`journal ...`, never prefixed with `sol` or `sol call`." + ) + assert openhands.SolAction.command.description == ( + "Single command-line invocation to run directly, without a shell: use " + "`sol`/`sol call ...` for normal journal access; run approved " + f"`journal` families ({family_list}) directly as " + "`journal ...`, never prefixed with `sol` or `sol call`." + ) + + def test_read_only_allowed_sol_call_returns_non_error_observation( fake_openhands, fixed_time, @@ -117,6 +139,54 @@ def test_read_only_policy_deny_is_recoverable_observation( assert events == [] +@pytest.mark.parametrize( + ("command", "reason"), + [ + ( + "sol call journal identity partner", + "policy_deny: `sol call journal identity` is not a `sol call` verb; " + "`identity` is an approved host command family — run it directly as " + "`journal identity partner`", + ), + ( + 'journal search "partner"', + "policy_deny: `journal search` is not a host command; run " + "`sol call journal search partner` instead", + ), + ( + "journal facet show work", + "policy_deny: `journal facet` is not a host command; run " + "`sol call journal facet show work` instead", + ), + ], +) +def test_policy_repair_denies_are_recoverable_without_running_command( + fake_openhands, + fixed_time, + tmp_path, + monkeypatch, + command, + reason, +): + events: list[dict] = [] + tool, executor = _sol_tool_and_executor( + tmp_path=tmp_path, + events=events, + ) + monkeypatch.setattr( + openhands, + "_run_command", + lambda _argv: pytest.fail("denied commands must not run"), + ) + + observation = tool(tool.action_from_arguments({"command": command})) + + assert observation.is_error is True + assert observation.text == reason + assert executor.read_call_count == 0 + assert events == [] + + def test_read_call_budget_overflow_emits_once_and_denies_recoverably( fake_openhands, fixed_time, diff --git a/tests/test_talent_cli.py b/tests/test_talent_cli.py index c291dee67..eaa6c1bf9 100644 --- a/tests/test_talent_cli.py +++ b/tests/test_talent_cli.py @@ -448,6 +448,23 @@ def test_scan_command_examples_dedupes_and_caps(): ] +def test_real_partner_prompt_uses_direct_journal_identity_without_contradiction(): + config = get_talent("partner") + + prompt_body, system_instruction = assemble_prompt(config, sol_tool_name="sol") + + assert system_instruction is not None + assert "approved host command families" in system_instruction + assert "journal ..." in system_instruction + assert "never prefixed with `sol` or `sol call`" in system_instruction + assert "no bare `journal ...` commands" not in system_instruction + assert "no other bare `journal ...` family" in system_instruction + assert "`journal identity partner` through the provided" in prompt_body + assert "`journal identity partner --update-section`" in prompt_body + assert "`sol`-surface read form" not in prompt_body + assert "there is no `sol call` verb" not in prompt_body + + def test_inventory_degrades_bad_access_tier_in_rows_table_and_json( tmp_path, monkeypatch,