From 5eed86fbbfda1ca3d93d14040db092dc242f5da4 Mon Sep 17 00:00:00 2001 From: dawn <90008@klbr.net> Date: Thu, 17 Sep 2026 10:28:52 +0300 Subject: [PATCH] feat(rlm): compose invariant child brief sections and warn on unlinted briefs --- .../.changes/subagent-brief-composition.md | 1 + packages/coding-agent/src/core/prompts/rlm.ts | 2 + .../coding-agent/test/system-prompt.test.ts | 2 + prime-agent-runtime/src/rlm/__init__.py | 172 +++++++++++++++++- prime-agent-runtime/test/test_repl.py | 2 +- .../test/test_skill_surface.py | 2 +- .../test/test_subagent_registry.py | 78 +++++++- 7 files changed, 254 insertions(+), 5 deletions(-) create mode 100644 packages/coding-agent/.changes/subagent-brief-composition.md diff --git a/packages/coding-agent/.changes/subagent-brief-composition.md b/packages/coding-agent/.changes/subagent-brief-composition.md new file mode 100644 index 000000000..ac02562a9 --- /dev/null +++ b/packages/coding-agent/.changes/subagent-brief-composition.md @@ -0,0 +1 @@ +- Added structured brief composition (`read_only`, `deliverable`, `evidence_bar`, `falsifier`) to `rlm.spawn` with unlinted brief warnings and invariant action contracts. diff --git a/packages/coding-agent/src/core/prompts/rlm.ts b/packages/coding-agent/src/core/prompts/rlm.ts index 40a14b3e8..82bd22547 100644 --- a/packages/coding-agent/src/core/prompts/rlm.ts +++ b/packages/coding-agent/src/core/prompts/rlm.ts @@ -257,6 +257,8 @@ export function buildSubagentGuidance( lines.push( "Fan-in results with `await rlm.collect(targets, timeout_ms=0)`: it returns typed snapshots of direct children (status, answer preview, error) without steering anyone; an explicit timeout blocks only that call until the children settle or the deadline passes.", "Large child outputs belong in files that you read selectively; `collect` snapshots are previews, not full results.", + "Children write deliverables atomically (write to `.partial` and rename into place).", + "The parent must re-run the child's one-command gate (a deterministic-output hash, a build provenance check, or the arithmetic) before any child-reported number enters memory, a report, or a message to the user.", "Delegate parallel context-heavy research or independent implementation; do a single known lookup, edit, or command inline.", ); if (options.includeRefineExamples ?? true) { diff --git a/packages/coding-agent/test/system-prompt.test.ts b/packages/coding-agent/test/system-prompt.test.ts index 22982b9cd..8281fd880 100644 --- a/packages/coding-agent/test/system-prompt.test.ts +++ b/packages/coding-agent/test/system-prompt.test.ts @@ -454,6 +454,8 @@ describe("buildSystemPrompt", () => { expect(prompt).toContain("session_dir"); expect(prompt).toContain("agent_observe"); expect(prompt).toContain("restricted to your parent, siblings, and direct children"); + expect(prompt).toContain("Children write deliverables atomically"); + expect(prompt).toContain("The parent must re-run the child's one-command gate"); }); test("omits ipython-only subagent guidance when ipython is inactive", () => { diff --git a/prime-agent-runtime/src/rlm/__init__.py b/prime-agent-runtime/src/rlm/__init__.py index 2c1929b15..56f4eff2c 100644 --- a/prime-agent-runtime/src/rlm/__init__.py +++ b/prime-agent-runtime/src/rlm/__init__.py @@ -2,6 +2,8 @@ from __future__ import annotations +import os +import re import sys import types from dataclasses import dataclass @@ -260,6 +262,106 @@ async def models() -> RLMModels: return RLMModels(payload) +_PATH_LIKE_PATTERN = re.compile( + r"(?:" + r"/(?:[a-zA-Z0-9_.-]+/)*[a-zA-Z0-9_.-]+" + r"|\./[a-zA-Z0-9_.-]+" + r"|[a-zA-Z0-9_.-]+/[a-zA-Z0-9_.-]+" + r"|[a-zA-Z0-9_.-]+\.(?:md|txt|json|yaml|yml|toml|py|ts|js|rs|go|diff|patch|log|csv)\b" + r")" +) + + +def _has_path_like_token(text: str) -> bool: + return bool(_PATH_LIKE_PATTERN.search(text)) + + +def _warn_brief_lint(name: str) -> None: + warning = ( + f"[WARNING] rlm.spawn: unlinted brief for child {name!r}: missing deliverable (or output path), " + "evidence_bar, and falsifier. Pass deliverable=..., evidence_bar=..., or falsifier=... " + "(see delegation guidance in model prompt)." + ) + try: + print(warning, file=sys.stderr) + except Exception: + pass + + +def _check_brief_lint( + prompt: str, + *, + name: str, + deliverable: str | None, + evidence_bar: str | None, + falsifier: str | None, +) -> None: + has_structured = bool(deliverable or evidence_bar or falsifier) + if not has_structured and not _has_path_like_token(prompt): + _warn_brief_lint(name) + + +def _current_child_depth() -> int: + try: + return int(os.environ.get("RLM_DEPTH", "0") or "0") + 1 + except ValueError: + return 1 + + +def compose_child_brief( + prompt: str, + *, + name: str, + depth: int | None = None, + read_only: bool = False, + deliverable: str | None = None, + evidence_bar: str | None = None, + falsifier: str | None = None, +) -> str: + """Compose the invariant and structured parts of a child task brief.""" + actual_depth = _current_child_depth() if depth is None else depth + sections: list[str] = [ + f"You are child subagent {name!r} at depth {actual_depth}.\n" + "Results arrive only through replies or files, never as an rlm.spawn() return value." + ] + + if read_only: + sections.append( + "# HARD RULES (READ-ONLY WORKER)\n" + f"- Do not modify anything outside your own scratch directory (/tmp/{name}).\n" + "- Do not create, edit, delete, move, or rename any file in the repository.\n" + "- Do not run state-changing git or jj commands (commit, reset, checkout, push, stash).\n" + "- No builds, GPU work, or package installs that change state.\n" + "- Reading and reasoning is the job. If changes are needed, describe them in the deliverable instead of making them." + ) + + clean_prompt = prompt.strip() + if clean_prompt.startswith("TASK:") or clean_prompt.startswith("# TASK"): + sections.append(clean_prompt) + else: + sections.append(f"TASK:\n{clean_prompt}") + + if deliverable is not None: + sections.append( + f"DELIVERABLE: {deliverable.strip()}\n" + f"Write to {deliverable.strip()}.partial and rename it into place when finished, " + "so a parent that reads the path early can never see a half-written artifact." + ) + + if evidence_bar is not None: + sections.append(f"EVIDENCE BAR:\n{evidence_bar.strip()}") + + if falsifier is not None: + sections.append(f"FALSIFIER:\n{falsifier.strip()}") + + sections.append( + f"FIRST ACTION: write /tmp/{name}/status.md with a 5-line plan, then update it after each step.\n" + "LAST ACTION: reply to the parent via `await agent_message.send(message, receiver_role='parent')`." + ) + + return "\n\n".join(sections) + + async def spawn( prompt: str, *, @@ -269,6 +371,10 @@ async def spawn( host: str | None = None, workspace: str | None = None, allow_push: bool = False, + read_only: bool = False, + deliverable: str | None = None, + evidence_bar: str | None = None, + falsifier: str | None = None, ) -> RLMSpawnHandle: """Spawn a recursive Prime Agent child and return once its task is admitted. @@ -290,6 +396,17 @@ async def spawn( ``allow_push=True`` authorizes the child's skills to publish (``git push``, force resets) that a child would otherwise refuse; the child inherits it as ``RLM_ALLOW_PUSH=1``. + + ``read_only=True`` adds a HARD RULES block restricting the child to its scratch + directory without modifying the repository, running state-changing git commands, + or executing stateful builds. + + ``deliverable`` specifies the destination file for the child's report, instructing + it to write to ``.partial`` and rename atomically into place. + + ``evidence_bar`` defines the standard of proof required for claims. + + ``falsifier`` defines the pre-registered condition that refutes the hypothesis. """ if not isinstance(prompt, str): raise TypeError(f"prompt must be str, got {type(prompt).__name__}") @@ -297,6 +414,29 @@ async def spawn( raise TypeError("name is required and must be a non-empty str") if not isinstance(model, str) or not model: raise TypeError("model is required and must be a non-empty str") + if not isinstance(read_only, bool): + raise TypeError(f"read_only must be a bool, got {type(read_only).__name__}") + if deliverable is not None and (not isinstance(deliverable, str) or not deliverable.strip()): + raise TypeError("deliverable must be a non-empty str when set") + if evidence_bar is not None and (not isinstance(evidence_bar, str) or not evidence_bar.strip()): + raise TypeError("evidence_bar must be a non-empty str when set") + if falsifier is not None and (not isinstance(falsifier, str) or not falsifier.strip()): + raise TypeError("falsifier must be a non-empty str when set") + _check_brief_lint( + prompt, + name=name, + deliverable=deliverable, + evidence_bar=evidence_bar, + falsifier=falsifier, + ) + composed_prompt = compose_child_brief( + prompt, + name=name, + read_only=read_only, + deliverable=deliverable, + evidence_bar=evidence_bar, + falsifier=falsifier, + ) kwargs: dict[str, Any] = {"name": name, "model": model} if thinking is not None: kwargs["thinking"] = thinking @@ -317,7 +457,7 @@ async def spawn( created = create_workspace(name) kwargs["cwd"] = str(created.path) try: - payload = await host_request("rlm.run", {"prompt": prompt, "kwargs": kwargs}) + payload = await host_request("rlm.run", {"prompt": composed_prompt, "kwargs": kwargs}) except BaseException: if created is not None: created.prune() @@ -698,6 +838,10 @@ class _RLMNamespace: host: str | None = None, workspace: str | None = None, allow_push: bool = False, + read_only: bool = False, + deliverable: str | None = None, + evidence_bar: str | None = None, + falsifier: str | None = None, ) -> RLMSpawnHandle: return await spawn( prompt, @@ -707,6 +851,31 @@ class _RLMNamespace: host=host, workspace=workspace, allow_push=allow_push, + read_only=read_only, + deliverable=deliverable, + evidence_bar=evidence_bar, + falsifier=falsifier, + ) + + def compose_child_brief( + self, + prompt: str, + *, + name: str, + depth: int | None = None, + read_only: bool = False, + deliverable: str | None = None, + evidence_bar: str | None = None, + falsifier: str | None = None, + ) -> str: + return compose_child_brief( + prompt, + name=name, + depth=depth, + read_only=read_only, + deliverable=deliverable, + evidence_bar=evidence_bar, + falsifier=falsifier, ) async def create_session( @@ -780,6 +949,7 @@ __all__ = [ "RLMSpawnHandle", "RLMSubagent", "RLMSubagentActivity", + "compose_child_brief", "create_session", "RefinementEvent", "bash", diff --git a/prime-agent-runtime/test/test_repl.py b/prime-agent-runtime/test/test_repl.py index 5847b93fc..8ed90f480 100644 --- a/prime-agent-runtime/test/test_repl.py +++ b/prime-agent-runtime/test/test_repl.py @@ -2541,7 +2541,7 @@ class ToolSurfaceTest(unittest.TestCase): events = self.repl.execute("s6", "rlm.spawn('x', name='n', role='coder')") text = self.traceback_text(events) self.assertEqual(one(events, "error")["ename"], "TypeError") - self.assertIn("signature: rlm.spawn(prompt, *, name, model, thinking?=None, host?=None, workspace?=None, allow_push?=False)", text) + self.assertIn("signature: rlm.spawn(prompt, *, name, model, thinking?=None, host?=None, workspace?=None, allow_push?=False, read_only?=False, deliverable?=None, evidence_bar?=None, falsifier?=None)", text) self.assertIn("nearest: 'role' -> model (rlm.spawn); use model=", text) def test_namespace_skill_surface_answers_for_a_wrapped_module(self): diff --git a/prime-agent-runtime/test/test_skill_surface.py b/prime-agent-runtime/test/test_skill_surface.py index e0784f4cf..044a6176e 100644 --- a/prime-agent-runtime/test/test_skill_surface.py +++ b/prime-agent-runtime/test/test_skill_surface.py @@ -256,7 +256,7 @@ def helper() -> str: self.assertIn("bash(command,", digest) self.assertIn("BashResult: ", digest) self.assertIn("BashHandle: ", digest) - self.assertIn("rlm.spawn(prompt, *, name, model, thinking?=None, host?=None, workspace?=None, allow_push?=False)", digest) + self.assertIn("rlm.spawn(prompt, *, name, model, thinking?=None, host?=None, workspace?=None, allow_push?=False, read_only?=False, deliverable?=None, evidence_bar?=None, falsifier?=None)", digest) class RemedyTest(unittest.TestCase): diff --git a/prime-agent-runtime/test/test_subagent_registry.py b/prime-agent-runtime/test/test_subagent_registry.py index 7ea6cc95a..fd0ae1839 100644 --- a/prime-agent-runtime/test/test_subagent_registry.py +++ b/prime-agent-runtime/test/test_subagent_registry.py @@ -95,10 +95,11 @@ class TestSubagentRegistry(unittest.TestCase): ) ) + expected_prompt = rlm_module.compose_child_brief("check the API", name="api-reviewer") host_request.assert_awaited_once_with( "rlm.run", { - "prompt": "check the API", + "prompt": expected_prompt, "kwargs": { "name": "api-reviewer", "model": "anthropic/claude-opus-5", @@ -127,9 +128,10 @@ class TestSubagentRegistry(unittest.TestCase): rlm_module.spawn("check the API", name="api-reviewer", model="openai-codex/gpt-5.6-terra") ) + expected_prompt = rlm_module.compose_child_brief("check the API", name="api-reviewer") host_request.assert_awaited_once_with( "rlm.run", - {"prompt": "check the API", "kwargs": {"name": "api-reviewer", "model": "openai-codex/gpt-5.6-terra"}}, + {"prompt": expected_prompt, "kwargs": {"name": "api-reviewer", "model": "openai-codex/gpt-5.6-terra"}}, ) self.assertEqual(result.model, "openai-codex/gpt-5.6-terra") @@ -490,5 +492,77 @@ class RlmSubagentExtrasTest(unittest.TestCase): asyncio.run(rlm_module.list_subagents()) +class TestSpawnBriefComposition(unittest.TestCase): + def test_composition_sections(self) -> None: + brief = rlm_module.compose_child_brief( + "run validation", + name="worker-1", + depth=1, + read_only=True, + deliverable="/tmp/report.md", + evidence_bar="commands required", + falsifier="speedup < 1", + ) + self.assertIn("You are child subagent 'worker-1' at depth 1.", brief) + self.assertIn("Results arrive only through replies or files", brief) + self.assertIn("# HARD RULES (READ-ONLY WORKER)", brief) + self.assertIn("TASK:\nrun validation", brief) + self.assertIn("DELIVERABLE: /tmp/report.md", brief) + self.assertIn("/tmp/report.md.partial", brief) + self.assertIn("EVIDENCE BAR:\ncommands required", brief) + self.assertIn("FALSIFIER:\nspeedup < 1", brief) + self.assertIn("FIRST ACTION: write /tmp/worker-1/status.md", brief) + self.assertIn("LAST ACTION: reply to the parent", brief) + + minimal = rlm_module.compose_child_brief("run validation", name="worker-2", depth=1) + self.assertNotIn("HARD RULES", minimal) + self.assertNotIn("DELIVERABLE", minimal) + self.assertNotIn("EVIDENCE BAR", minimal) + self.assertNotIn("FALSIFIER", minimal) + + def test_lint_warning_behavior(self) -> None: + import contextlib + import io + + cases = [ + ("plain prompt without path", None, None, None, True), + ("plain prompt", "/tmp/out.md", None, None, False), + ("plain prompt", None, "evidence", None, False), + ("plain prompt", None, None, "falsifier", False), + ("inspect src/index.ts", None, None, None, False), + ("check /tmp/foo", None, None, None, False), + ] + for prompt, d, e, f, should_warn in cases: + buf = io.StringIO() + with contextlib.redirect_stderr(buf): + rlm_module._check_brief_lint(prompt, name="w", deliverable=d, evidence_bar=e, falsifier=f) + has_warning = "[WARNING] rlm.spawn: unlinted brief" in buf.getvalue() + self.assertEqual(has_warning, should_warn, f"failed for {prompt=}") + + def test_spawn_passes_composed_prompt(self) -> None: + host_request = AsyncMock( + return_value={ + "rlm_child_id": "sub-c1", + "name": "scout", + "session_dir": "/tmp/scout", + "model": "m", + } + ) + with patch.object(rlm_module, "host_request", host_request): + asyncio.run( + rlm_module.spawn( + "profile memory", + name="scout", + model="m", + read_only=True, + deliverable="/tmp/mem.md", + ) + ) + sent = host_request.await_args.args[1]["prompt"] + self.assertIn("# HARD RULES", sent) + self.assertIn("DELIVERABLE: /tmp/mem.md", sent) + self.assertIn("/tmp/mem.md.partial", sent) + + if __name__ == "__main__": unittest.main() -- 2.51.2