diff --git a/src/context.rs b/src/context.rs index fc57799..826891f 100644 --- a/src/context.rs +++ b/src/context.rs @@ -1,9 +1,9 @@ -//! The context stack: builds a role's system prompt each turn from -//! a base fragment, layered `system.md` fragments, and `context: true` -//! entries. +//! The context stack: builds a role's system prompt from a base +//! fragment, layered `system.md` fragments, and `context: true` entries. use std::fs; use std::path::PathBuf; +use std::sync::Arc; use crate::knowledge::Mount; @@ -14,29 +14,60 @@ const DIVIDER: &str = "\n\n---\n\n"; /// The heading every context entry gets when pushed into the prompt. const ENTRY_HEADING: &str = "### "; -/// A system prompt built from layer fragments and context entries. +/// The blank line that separates one context entry from the next +/// entry's heading. +const ENTRY_SEPARATOR: &str = "\n\n"; + +/// What a system prompt is composed from: a base fragment, the layer +/// directories that may each add a `system.md`, and the mounted +/// knowledge. +/// +/// [`ContextStack::prompt`] composes the prompt every time it is called, +/// and the DM calls it at the start of every turn. The two halves age +/// differently. Each layer's `system.md` is read from disk on every +/// call, so an edit to a fragment reaches the next turn. The entries +/// are the ones the mount holds, and the mount is a snapshot taken when +/// the session started, so a file added to a layer mid-session is not +/// mounted and does not reach the prompt. #[derive(Debug)] pub struct ContextStack { - /// The full assembled system prompt, ready for the model. - prompt: String, + base: String, + layers: Vec, + mount: Arc, } impl ContextStack { - /// Builds the context stack from a `base` fragment, `layers` (lowest - /// first), and `mount`. The base fragment is always first; then each - /// filesystem layer's `system.md` (if it exists); then every mounted - /// entry with `context: true`. + /// Builds a context stack over a `base` fragment, `layers` (lowest + /// first), and `mount`. /// /// The base fragment is what distinguishes roles: the DM base sets /// identity, narration style, and tool usage. A planner agent would /// pass a different base. What stays the same is how layers and /// context entries are assembled on top of it. - pub fn build(base: &str, layers: &[PathBuf], mount: &Mount) -> Result { + pub fn new(base: &str, layers: &[PathBuf], mount: Arc) -> Self { + Self { + base: base.to_string(), + layers: layers.to_vec(), + mount, + } + } + + /// Composes the system prompt, without a trailing newline: the base + /// fragment first, then each layer's `system.md` in mount order, + /// then the mount's knowledge directories, then every mounted entry + /// with `context: true`, each separated from the next by a blank + /// line. + /// + /// A layer with no `system.md` contributes nothing and is not a + /// failure. A layer whose `system.md` cannot be read is: the caller + /// turns that into a failed turn rather than sending the model a + /// prompt that is quietly missing a layer. + pub fn prompt(&self) -> Result { let mut prompt = String::new(); - prompt.push_str(base.trim()); + prompt.push_str(self.base.trim()); - for layer_root in layers { + for layer_root in &self.layers { let fragment_path = layer_root.join("system.md"); match fs::read_to_string(&fragment_path) { Ok(text) => { @@ -55,7 +86,7 @@ impl ContextStack { } } - let top_dirs = top_level_directories(mount); + let top_dirs = top_level_directories(&self.mount); if !top_dirs.is_empty() { prompt.push_str(DIVIDER); prompt.push_str("## Knowledge directories\n\n"); @@ -63,7 +94,7 @@ impl ContextStack { prompt.push_str(&top_dirs); } - let context_entries: Vec<_> = mount.entries().filter(|e| e.context).collect(); + let context_entries: Vec<_> = self.mount.entries().filter(|e| e.context).collect(); if !context_entries.is_empty() { if !prompt.is_empty() { prompt.push_str(DIVIDER); @@ -75,17 +106,11 @@ impl ContextStack { prompt.push_str(&entry.name); prompt.push('\n'); prompt.push_str(entry.body.trim()); + prompt.push_str(ENTRY_SEPARATOR); } } - Ok(Self { - prompt: prompt.trim().to_string(), - }) - } - - /// The full system prompt, without a trailing newline. - pub fn prompt(&self) -> &str { - &self.prompt + Ok(prompt.trim().to_string()) } } diff --git a/src/context_tests.rs b/src/context_tests.rs index bc53cbc..c88d1eb 100644 --- a/src/context_tests.rs +++ b/src/context_tests.rs @@ -1,9 +1,11 @@ use super::*; use crate::knowledge::fixtures::{EntrySpec, Layer, mount_from_specs}; -fn build(layers: &[&Layer], mount: &Mount) -> Result { +/// Composes the prompt over `layers` and `mount`, the way the DM does at +/// the start of a turn. +fn build(layers: &[&Layer], mount: Mount) -> Result { let roots: Vec = layers.iter().map(|layer| layer.path()).collect(); - ContextStack::build(TEST_BASE, &roots, mount) + ContextStack::new(TEST_BASE, &roots, Arc::new(mount)).prompt() } const TEST_BASE: &str = "You are a test DM."; @@ -19,9 +21,9 @@ fn the_base_layer_is_always_the_first_part_of_the_prompt() { let mount = mount_from_specs(&[]); let layer = Layer::empty(); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().starts_with(base())); + assert!(prompt.starts_with(base())); } #[test] @@ -30,10 +32,10 @@ fn a_single_layer_fragment_appends_after_the_base() { layer.write("system.md", "System-specific rules."); let mount = mount_from_specs(&[]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().starts_with(base())); - assert!(stack.prompt().ends_with("System-specific rules.")); + assert!(prompt.starts_with(base())); + assert!(prompt.ends_with("System-specific rules.")); } #[test] @@ -44,15 +46,15 @@ fn two_layers_fragments_are_concatenated_lowest_first_after_the_base() { high.write("system.md", "Second."); let mount = mount_from_specs(&[]); - let stack = build(&[&low, &high], &mount).unwrap(); + let prompt = build(&[&low, &high], mount).unwrap(); // Base comes first, then the two fragments in order. - let base_pos = stack.prompt().find(base()).unwrap(); - let first_pos = stack.prompt().find("First.").unwrap(); - let second_pos = stack.prompt().find("Second.").unwrap(); + let base_pos = prompt.find(base()).unwrap(); + let first_pos = prompt.find("First.").unwrap(); + let second_pos = prompt.find("Second.").unwrap(); assert!(base_pos < first_pos); assert!(first_pos < second_pos); - assert!(stack.prompt().ends_with("Second.")); + assert!(prompt.ends_with("Second.")); } #[test] @@ -61,10 +63,10 @@ fn a_missing_system_md_is_not_an_error() { // No system.md written. let mount = mount_from_specs(&[]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); // The base layer is still present. - assert!(stack.prompt().starts_with(base())); + assert!(prompt.starts_with(base())); } #[test] @@ -73,9 +75,21 @@ fn fragments_are_trimmed_before_concatenation() { layer.write("system.md", "\n\n Padded text. \n\n"); let mount = mount_from_specs(&[]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().contains("\n\nPadded text.")); + assert!(prompt.contains("\n\nPadded text.")); +} + +#[test] +fn a_fragment_edited_after_the_stack_is_built_reaches_the_next_prompt() { + let layer = Layer::empty(); + layer.write("system.md", "The first draft."); + let stack = ContextStack::new(TEST_BASE, &[layer.path()], Arc::new(mount_from_specs(&[]))); + assert!(stack.prompt().unwrap().ends_with("The first draft.")); + + layer.write("system.md", "The second draft."); + + assert!(stack.prompt().unwrap().ends_with("The second draft.")); } #[test] @@ -85,13 +99,9 @@ fn a_context_entry_appears_under_its_address_heading() { .with_context(true) .with_body("A warm tavern on the harbor.")]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!( - stack - .prompt() - .contains("### lore/places/inn — The Tipsy Fox\nA warm tavern on the harbor.") - ); + assert!(prompt.contains("### lore/places/inn — The Tipsy Fox\nA warm tavern on the harbor.")); } #[test] @@ -106,10 +116,15 @@ fn multiple_context_entries_are_all_included() { .with_body("Move twice your speed."), ]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().contains("The Inn")); - assert!(stack.prompt().contains("Dash")); + // Each entry ends with a blank line, so the next entry's heading + // starts a block of its own instead of running on from the last + // line of body text. + assert!(prompt.ends_with( + "### lore/places/inn — The Inn\nCozy.\n\n\ + ### rules/actions/dash — Dash\nMove twice your speed." + )); } #[test] @@ -122,10 +137,10 @@ fn entries_without_context_flag_are_skipped() { .with_body("Wet."), ]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(!stack.prompt().contains("The Inn")); - assert!(stack.prompt().contains("The Dock")); + assert!(!prompt.contains("The Inn")); + assert!(prompt.contains("The Dock")); } #[test] @@ -136,10 +151,10 @@ fn fragments_and_entries_are_separated() { .with_context(true) .with_body("Warm.")]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().contains("System fragment.")); - assert!(stack.prompt().contains("### lore/places/inn")); + assert!(prompt.contains("System fragment.")); + assert!(prompt.contains("### lore/places/inn")); } #[test] @@ -149,10 +164,10 @@ fn entry_body_is_trimmed() { .with_context(true) .with_body("\n\n Warm. \n\n")]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().contains("Warm.")); - assert!(!stack.prompt().contains(" Warm. ")); + assert!(prompt.contains("Warm.")); + assert!(!prompt.contains(" Warm. ")); } #[test] @@ -163,7 +178,7 @@ fn an_unreadable_system_md_fails() { std::fs::write(&path, [0xFF, 0xFE]).unwrap(); let mount = mount_from_specs(&[]); - let error = build(&[&layer], &mount).unwrap_err(); + let error = build(&[&layer], mount).unwrap_err(); assert!(error.contains("system.md")); } @@ -178,18 +193,14 @@ fn top_level_directories_are_listed_with_frontmatter_keys() { EntrySpec::new("lore/people/martha", "Martha"), ]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().contains("## Knowledge directories")); + assert!(prompt.contains("## Knowledge directories")); // Scopes with frontmatter keys show them. - assert!(stack.prompt().contains("- `rules/monsters` — type")); - assert!(stack.prompt().contains("- `rules/spells` — type")); + assert!(prompt.contains("- `rules/monsters` — type")); + assert!(prompt.contains("- `rules/spells` — type")); // Scopes with only `name` show just the scope, with no key list. - assert!( - stack - .prompt() - .contains("- `lore/people`\n- `lore/places`\n") - ); + assert!(prompt.contains("- `lore/people`\n- `lore/places`\n")); } #[test] @@ -199,7 +210,7 @@ fn a_non_string_frontmatter_key_is_left_out_of_the_directory_listing() { .with_kind("spell") .with_field("3", "third level")]); - let stack = build(&[&layer], &mount).unwrap(); + let prompt = build(&[&layer], mount).unwrap(); - assert!(stack.prompt().contains("- `rules/spells` — type")); + assert!(prompt.contains("- `rules/spells` — type")); } diff --git a/src/dm/base-system.md b/src/dm/base-system.md index 502563a..d44e202 100644 --- a/src/dm/base-system.md +++ b/src/dm/base-system.md @@ -12,15 +12,15 @@ When you need all entries with a particular frontmatter value — every 9th-leve Knowledge comes in layers, lowest first. The bottom layer is the rules system. The layers above it add module content, the world the DM built, and the player's own tuning. A higher layer's entry replaces a lower one at the same address. Some entries are pushed into this prompt every turn; you always know them without looking them up. -When a rule matters, look it up instead of guessing. - ## Visibility -Every tool call takes a `visibility` argument. Knowledge entries can also carry a `visibility` field in their frontmatter. +Every tool call takes a `visibility` argument. Knowledge entries can also carry a `visibility` field in their frontmatter. That field limits what the player sees when a `lookup` or a `read` finds the entry. - `public`: the player sees the full result. Use this when a player at a real table would see the dice or read the page. -- `screened`: the player sees that something happened behind the screen, but not what. Use this when the player should know a roll or a lookup occurred without seeing its numbers. A screened knowledge entry still reaches your prompt, but the player never sees it referenced. -- `secret`: the player sees nothing. Use this for rolls and knowledge the player should not know about at all. A secret entry still goes into your prompt when it is pushed, because the DM knows the innkeeper is a retired assassin even if the player does not. +- `screened`: the player sees that something happened behind the screen, but not what. Use this when the player should know a roll or a lookup occurred without seeing its numbers. A `lookup` or a `read` that finds a screened entry shows the screened line even when you ask for `public`. +- `secret`: the player sees nothing. Use this for rolls and knowledge the player should not know about at all. A `lookup` or a `read` that finds a secret entry shows nothing at all, whatever you ask for. + +You always get the full text of every entry, screened and secret alike. A screened or secret entry still goes into this prompt when it is pushed: the DM knows the innkeeper is a retired assassin even if the player does not. The engine decides what the transcript shows for a tool call. It does not read your narration. Keep a screened or a secret entry out of what you say until the story reveals it. ## Tools diff --git a/src/dm/dm_tests.rs b/src/dm/dm_tests.rs index a52c71d..055df46 100644 --- a/src/dm/dm_tests.rs +++ b/src/dm/dm_tests.rs @@ -106,6 +106,7 @@ fn dm_for(api_base: String) -> Dm { &[], None, ) + .unwrap() } /// A `Dm` whose toolbox is the full default: dice, knowledge, and the @@ -121,6 +122,7 @@ fn dm_with_campaign(api_base: String, world: &TempDir) -> Dm { &[], Some(Campaign::open(world.path()).unwrap()), ) + .unwrap() } /// Discards every event and keeps streaming. Pass this to `turn` in @@ -222,7 +224,8 @@ fn a_pre_turn_campaign_write_failure_ends_the_turn_as_an_error_before_any_reques Arc::new(fixtures::mount(&[])), &[], Some(campaign), - ); + ) + .unwrap(); let error = dm.turn("I sleep.", &mut ignore_event).unwrap_err(); @@ -373,3 +376,72 @@ fn a_cancelled_turns_history_is_unchanged_for_the_next_request() { assert_eq!(messages.len(), 2); assert_eq!(messages[1]["content"], "a fresh input"); } + +/// A `Dm` over `layers`, so a test can edit a layer's `system.md` while +/// the session runs. +fn dm_over(api_base: String, layers: &[PathBuf]) -> Result { + Dm::new( + Config { + api_base, + api_key: "sk-test".to_string(), + model: "gpt-4o-mini".to_string(), + }, + Arc::new(fixtures::mount(&[])), + layers, + None, + ) +} + +#[test] +fn a_layer_that_cannot_be_read_fails_the_dm_at_startup() { + let layer = fixtures::Layer::empty(); + fs::write(layer.path().join("system.md"), [0xFF, 0xFE]).unwrap(); + + let error = dm_over("http://127.0.0.1:0".to_string(), &[layer.path()]) + .err() + .expect("a layer that cannot be read must fail the dm"); + + assert!(error.contains("system.md")); +} + +#[test] +fn each_turn_composes_the_system_prompt_again_from_the_layers_on_disk() { + let layer = fixtures::Layer::empty(); + layer.write("system.md", "The first draft."); + let (url, requests, server) = fake_server(vec![ + sse_response("data: [DONE]\n\n"), + sse_response("data: [DONE]\n\n"), + ]); + let mut dm = dm_over(url, &[layer.path()]).unwrap(); + + dm.turn("I open the door.", &mut ignore_event).unwrap(); + layer.write("system.md", "The second draft."); + dm.turn("I step inside.", &mut ignore_event).unwrap(); + + server.join().unwrap(); + let first = sent_messages(&requests.recv().unwrap()); + let second = sent_messages(&requests.recv().unwrap()); + assert!( + first[0]["content"] + .as_str() + .unwrap() + .ends_with("The first draft.") + ); + assert!( + second[0]["content"] + .as_str() + .unwrap() + .ends_with("The second draft.") + ); +} + +#[test] +fn a_layer_that_goes_unreadable_mid_session_ends_the_turn_as_an_error() { + let layer = fixtures::Layer::empty(); + let mut dm = dm_over("http://127.0.0.1:0".to_string(), &[layer.path()]).unwrap(); + fs::write(layer.path().join("system.md"), [0xFF, 0xFE]).unwrap(); + + let error = dm.turn("I open the door.", &mut ignore_event).unwrap_err(); + + assert!(error.to_string().contains("system.md")); +} diff --git a/src/dm/dm_tool_round_tests.rs b/src/dm/dm_tool_round_tests.rs index f5a67bf..3bd07e1 100644 --- a/src/dm/dm_tool_round_tests.rs +++ b/src/dm/dm_tool_round_tests.rs @@ -14,9 +14,8 @@ use rand::rngs::StdRng; use serde_json::json; use tempfile::TempDir; -fn empty_context() -> ContextStack { - let mount = fixtures::mount(&[]); - ContextStack::build("", &[], &mount).unwrap() +fn empty_context() -> Arc { + Arc::new(ContextStack::new("", &[], Arc::new(fixtures::mount(&[])))) } /// Builds a `Dm` whose toolbox holds a single dice tool seeded from @@ -34,6 +33,7 @@ fn dm_with_seeded_dice(api_base: String, seed: u64) -> Dm { empty_context(), None, ) + .unwrap() } /// Parses a captured request body as JSON. @@ -191,7 +191,8 @@ fn a_post_turn_transcript_write_failure_appends_a_warning_instead_of_failing_the Arc::new(fixtures::mount(&[])), &[], Some(campaign), - ); + ) + .unwrap(); let mut deltas = Vec::new(); let turn = dm diff --git a/src/dm/mod.rs b/src/dm/mod.rs index aa14ee3..843f4ab 100644 --- a/src/dm/mod.rs +++ b/src/dm/mod.rs @@ -24,6 +24,7 @@ use tools::mark::MarkTool; use tools::read::ReadTool; use tools::recall::RecallTool; +mod report; pub mod tools; /// Told to the model in place of its tools on a withheld round: it has @@ -64,14 +65,18 @@ pub enum Turn { Cancelled(String), } -/// Why a turn ended without a reply: the chat request failed, or the -/// player's line could not be recorded before anything streamed. +/// Why a turn ended without a reply: the system prompt could not be +/// composed, the chat request failed, or the player's line could not be +/// recorded before anything streamed. #[derive(Debug)] pub enum TurnError { /// Sending the request or reading the streamed response failed. Chat(ChatError), /// The campaign could not record the turn. Campaign(String), + /// A layer's `system.md` could not be read, so this turn has no + /// system prompt to send. + Context(String), } impl fmt::Display for TurnError { @@ -79,6 +84,7 @@ impl fmt::Display for TurnError { match self { TurnError::Chat(error) => write!(f, "{error}"), TurnError::Campaign(message) => write!(f, "{message}"), + TurnError::Context(message) => write!(f, "{message}"), } } } @@ -109,7 +115,7 @@ pub struct Dm { client: Client, toolbox: Toolbox, campaign: Option, - context: ContextStack, + context: Arc, history: Vec, } @@ -117,12 +123,14 @@ impl Dm { /// Builds a `Dm` from `config`, `mount`, `layers` (lowest first), /// and `campaign`, seeding the history with the system prompt and the /// toolbox with the dice tool, the two knowledge tools (`lookup` and - /// `read`), and, when a `campaign` is given, the two history tools - /// (`mark` and `recall`). + /// `read`), the `context` tool, and, when a `campaign` is given, the + /// two history tools (`mark` and `recall`). /// - /// The system prompt is built from `layers`' `system.md` fragments - /// and the mount's `context: true` entries, assembled fresh here and - /// rebuilt every turn. + /// The system prompt comes from `layers`' `system.md` fragments and + /// the mount's `context: true` entries. `turn` composes it again + /// every turn, and this builds it once here so a layer that cannot + /// be read fails the session at startup rather than on the first + /// thing the player types. /// /// The dice tool rolls with a `StdRng` seeded from `rand::make_rng` /// rather than the thread-local `rand::rng()` directly: a `Dm` moves @@ -137,7 +145,7 @@ impl Dm { mount: Arc, layers: &[PathBuf], campaign: Option, - ) -> Self { + ) -> Result { let rng: StdRng = rand::make_rng(); let mut tools: Vec> = vec![ Box::new(DiceTool::new(rng)), @@ -148,9 +156,12 @@ impl Dm { tools.push(Box::new(MarkTool::new(campaign.clone()))); tools.push(Box::new(RecallTool::new(campaign))); } - let context = ContextStack::build(include_str!("base-system.md"), layers, &mount) - .expect("context stack must build"); - tools.push(Box::new(ContextTool::new(context.prompt().to_string()))); + let context = Arc::new(ContextStack::new( + include_str!("base-system.md"), + layers, + mount, + )); + tools.push(Box::new(ContextTool::new(Arc::clone(&context)))); let toolbox = Toolbox::new(tools); Self::with_toolbox(config, toolbox, context, campaign) } @@ -159,25 +170,38 @@ impl Dm { /// callers that need a toolbox other than the default, such as a test /// with a seeded dice tool or a fake one. `campaign` is where /// narration is recorded; a `None` records nothing. + /// + /// Fails when `context` cannot compose its first prompt. pub fn with_toolbox( config: Config, toolbox: Toolbox, - context: ContextStack, + context: Arc, campaign: Option, - ) -> Self { + ) -> Result { let client = Client { api_base: config.api_base, api_key: config.api_key, model: config.model, }; - let history = vec![system_message(context.prompt())]; - Self { + let history = vec![system_message(&context.prompt()?)]; + Ok(Self { client, toolbox, campaign, context, history, - } + }) + } + + /// The `/context` report: the system prompt as the context stack + /// composes it now, and every tool the model can call, rendered for + /// a player who asks what the DM was told. + pub fn context_report(&self) -> String { + let prompt = self + .context + .prompt() + .unwrap_or_else(|error| format!("(the system prompt cannot be built: {error})")); + report::render(&prompt, &self.toolbox.definitions()) } /// Runs a turn from `input` as a loop of rounds, streaming narration @@ -218,15 +242,20 @@ impl Dm { /// after the reply has already streamed, so a failure there cannot /// undo what the player saw; it turns into a warning appended to the /// reply and streamed through `on_delta` like any other text. + /// + /// Composing the system prompt happens first of all, so a layer + /// whose `system.md` went unreadable mid-session ends the turn as an + /// error with the history untouched. pub fn turn( &mut self, input: &str, on_delta: &mut dyn FnMut(TurnDelta) -> ControlFlow<()>, ) -> Result { - // The system prompt is built fresh: if the mount has changed - // since the last turn, the next request carries the latest - // fragments and context entries. - self.history[0] = system_message(self.context.prompt()); + // Each layer's system.md is read from disk again here, so an + // edit to a fragment reaches this turn. The context entries come + // from the mount, which stays as it was scanned at startup. + let prompt = self.context.prompt().map_err(TurnError::Context)?; + self.history[0] = system_message(&prompt); let user_message = Message { role: Role::User, diff --git a/src/dm/report.rs b/src/dm/report.rs new file mode 100644 index 0000000..b07776b --- /dev/null +++ b/src/dm/report.rs @@ -0,0 +1,73 @@ +//! The `/context` report: everything the DM was told, written out for +//! the player. +//! +//! The report is the system prompt the context stack composes, then one +//! block per tool the toolbox declares, then the size of the whole +//! thing. It renders from the running `Dm`'s own prompt and toolbox, so +//! what the player reads is what the model was sent. + +use serde_json::Value; + +/// Shown in place of a field a declaration left out. +const MISSING: &str = "?"; + +/// How many characters of English prose one token is worth. +/// +/// Most tokenizers average 3.5 to 4 characters per token, so 4 makes the +/// token count a conservative estimate rather than an alarming one. +const CHARS_PER_TOKEN: usize = 4; + +/// Renders `prompt` and `definitions` as the `/context` report. +pub(super) fn render(prompt: &str, definitions: &[Value]) -> String { + let tools: Vec = definitions.iter().map(tool_block).collect(); + let report = format!("{prompt}\n\n---\n\n## Tools\n\n{}", tools.join("\n\n")); + let chars = report.len(); + let tokens = chars / CHARS_PER_TOKEN; + format!("{report}\n\n---\n\n*{chars} chars, ~{tokens} tokens*") +} + +/// One tool's block: its name, the description the model reads, and +/// every parameter it declares, in alphabetical order with the required +/// ones named at the end. +fn tool_block(definition: &Value) -> String { + let function = &definition["function"]; + let name = function["name"].as_str().unwrap_or(MISSING); + let description = function["description"].as_str().unwrap_or_default(); + let parameters = parameter_lines(&function["parameters"]["properties"]); + let required = required_line(&function["parameters"]["required"]); + format!("### {name}\n\n{description}\n\nParameters:\n{parameters}{required}") +} + +/// One line per declared parameter, sorted, so a report of the same +/// toolbox always reads the same way. +fn parameter_lines(properties: &Value) -> String { + let Some(properties) = properties.as_object() else { + return String::new(); + }; + let mut lines: Vec = properties + .iter() + .map(|(key, parameter)| { + let kind = parameter["type"].as_str().unwrap_or(MISSING); + let description = parameter["description"].as_str().unwrap_or_default(); + format!(" - `{key}` ({kind}): {description}") + }) + .collect(); + lines.sort(); + lines.join("\n") +} + +/// The names a tool requires, or nothing at all when it requires none. +fn required_line(required: &Value) -> String { + let names: Vec<&str> = required + .as_array() + .map(|names| names.iter().filter_map(Value::as_str).collect()) + .unwrap_or_default(); + if names.is_empty() { + return String::new(); + } + format!("\n required: {}", names.join(", ")) +} + +#[cfg(test)] +#[path = "report_tests.rs"] +mod tests; diff --git a/src/dm/report_tests.rs b/src/dm/report_tests.rs new file mode 100644 index 0000000..e41ad2b --- /dev/null +++ b/src/dm/report_tests.rs @@ -0,0 +1,163 @@ +//! Tests for `report.rs`: the pure rendering, and the report a real +//! `Dm` produces from its own prompt and toolbox. + +use super::*; +use crate::campaign::Campaign; +use crate::config::Config; +use crate::dm::Dm; +use crate::knowledge::fixtures::{self, Layer}; +use serde_json::json; +use std::path::PathBuf; +use std::sync::Arc; +use tempfile::TempDir; + +/// A declaration in the shape `Toolbox::definitions` produces. +fn definition(name: &str, description: &str, properties: Value, required: Value) -> Value { + json!({ + "type": "function", + "function": { + "name": name, + "description": description, + "parameters": { + "type": "object", + "properties": properties, + "required": required, + }, + }, + }) +} + +/// A `Dm` over `layers` with the full six-tool toolbox, pointed at an +/// api base nothing ever calls: the report needs no requests. +fn dm(layers: &[PathBuf], world: &TempDir) -> Dm { + Dm::new( + Config { + api_base: "http://127.0.0.1:0".to_string(), + api_key: "sk-test".to_string(), + model: "gpt-4o-mini".to_string(), + }, + Arc::new(fixtures::mount(&[])), + layers, + Some(Campaign::open(world.path()).unwrap()), + ) + .unwrap() +} + +/// Every `### ` heading in `report`, which is one per tool. +fn tool_names(report: &str) -> Vec<&str> { + report + .lines() + .filter_map(|line| line.strip_prefix("### ")) + .collect() +} + +#[test] +fn the_report_opens_with_the_system_prompt() { + let report = render("You are the DM.", &[]); + + assert!(report.starts_with("You are the DM.\n\n---\n\n## Tools\n\n")); +} + +#[test] +fn a_tool_block_names_the_tool_and_reads_out_its_description() { + let report = render( + "prompt", + &[definition("roll", "Roll dice.", json!({}), json!([]))], + ); + + assert!(report.contains("### roll\n\nRoll dice.\n\nParameters:\n")); +} + +#[test] +fn parameters_are_listed_in_alphabetical_order_with_the_required_ones_named() { + let block = tool_block(&definition( + "roll", + "Roll dice.", + json!({ + "notation": {"type": "string", "description": "What to roll."}, + "aim": {"type": "string", "description": "Where to aim."}, + }), + json!(["notation", "aim"]), + )); + + assert!(block.ends_with( + "Parameters:\n\ + \x20 - `aim` (string): Where to aim.\n\ + \x20 - `notation` (string): What to roll.\n\ + \x20 required: notation, aim" + )); +} + +#[test] +fn a_tool_with_no_required_parameters_has_no_required_line() { + let block = tool_block(&definition("context", "Show it.", json!({}), json!([]))); + + assert!(block.ends_with("Parameters:\n")); +} + +#[test] +fn a_declaration_missing_its_fields_renders_placeholders() { + let block = tool_block(&json!({ + "function": {"parameters": {"properties": {"target": {}}}} + })); + + assert_eq!(block, "### ?\n\n\n\nParameters:\n - `target` (?): "); +} + +#[test] +fn a_declaration_with_no_parameters_at_all_lists_none() { + let block = tool_block(&json!({ + "function": {"name": "bare", "description": "No arguments."} + })); + + assert_eq!(block, "### bare\n\nNo arguments.\n\nParameters:\n"); +} + +#[test] +fn the_report_ends_with_its_own_size_in_characters_and_estimated_tokens() { + // The 23 characters are the prompt, the divider, and the empty + // tools section; the size line itself is not counted. + assert_eq!( + render("prompt", &[]), + "prompt\n\n---\n\n## Tools\n\n\n\n---\n\n*23 chars, ~5 tokens*" + ); +} + +#[test] +fn the_report_lists_exactly_the_six_tools_the_dm_can_call() { + let world = TempDir::new().unwrap(); + + let report = dm(&[], &world).context_report(); + + assert_eq!( + tool_names(&report), + ["roll", "lookup", "read", "mark", "recall", "context"] + ); +} + +#[test] +fn the_report_carries_the_system_prompt_the_dm_would_send() { + let world = TempDir::new().unwrap(); + let layer = Layer::empty(); + layer.write("system.md", "The house rules."); + + let report = dm(&[layer.path()], &world).context_report(); + + assert!(report.contains("The house rules.")); +} + +#[test] +fn a_prompt_that_cannot_be_composed_says_so_in_place_of_the_prompt() { + let world = TempDir::new().unwrap(); + let layer = Layer::empty(); + let dm = dm(&[layer.path()], &world); + // The fragment appears only after the session starts, so the DM + // built its first prompt without it and the report is the first to + // meet it. + std::fs::write(layer.path().join("system.md"), [0xFF, 0xFE]).unwrap(); + + let report = dm.context_report(); + + assert!(report.contains("the system prompt cannot be built")); + assert!(report.contains("system.md")); +} diff --git a/src/dm/tools/context.md b/src/dm/tools/context.md new file mode 100644 index 0000000..586499e --- /dev/null +++ b/src/dm/tools/context.md @@ -0,0 +1,3 @@ +Reveal the DM's full system prompt, exactly as the context stack built it for this turn. It takes no arguments of its own; call it with `visibility` alone. + +The prompt holds screened and secret knowledge, so the player never sees it. This tool renders at `screened` at most, whatever `visibility` you pass: the transcript says the prompt was revealed and how long it is, and nothing else. Pass `secret` to leave even that out. diff --git a/src/dm/tools/context.rs b/src/dm/tools/context.rs index 5fa6aab..ba44029 100644 --- a/src/dm/tools/context.rs +++ b/src/dm/tools/context.rs @@ -1,24 +1,30 @@ -//! The `context` debug tool: shows the DM's current system prompt in -//! the transcript and to the model. +//! The `context` debug tool: shows the DM's current system prompt to +//! the model, and how long that prompt is to the player. + +use std::sync::Arc; use ratatui::style::{Modifier, Style}; use ratatui::text::{Line, Span, Text}; use serde_json::{Value, json}; +use crate::context::ContextStack; + use super::{Tool, ToolReply, Visibility}; +const CONTEXT_MD: &str = include_str!("context.md"); + /// What the transcript shows when the DM reveals the context. const LABEL: &str = "🔍 context"; /// The `context` tool: reveals the DM's full system prompt. pub struct ContextTool { - prompt: String, + context: Arc, } impl ContextTool { - /// Builds a context tool that reveals `prompt`. - pub fn new(prompt: String) -> Self { - Self { prompt } + /// Builds a context tool that reveals the prompt `context` composes. + pub fn new(context: Arc) -> Self { + Self { context } } } @@ -32,7 +38,7 @@ impl Tool for ContextTool { "type": "function", "function": { "name": "context", - "description": "Reveal the DM's full system prompt, exactly as it was built this turn. No arguments — call it empty.", + "description": CONTEXT_MD.trim_end(), "parameters": { "type": "object", "properties": {}, @@ -41,18 +47,28 @@ impl Tool for ContextTool { }) } + /// Composes the prompt the same way the turn did, so the model reads + /// what it was actually told rather than a copy taken at startup. A + /// layer whose `system.md` cannot be read becomes this call's error + /// text, which the model reads and the player does not. + /// + /// The cap is `Screened`, because the prompt carries every screened + /// and secret entry the mount pushed into it. The player learns that + /// the DM looked, and how long the prompt is, and no more. Both + /// transcript lines say the same thing: with the cap in place a + /// public line holding the prompt itself could never render, and it + /// would leak the whole prompt the day someone lifted the cap. fn call(&mut self, _args: &Value, _visibility: Visibility) -> Result { - let display = format!("{}\n{}", LABEL, self.prompt); + let prompt = self.context.prompt()?; + let line = Text::from(Line::from(Span::styled( + format!("{LABEL} ({} chars)", prompt.len()), + dim_style(), + ))); Ok(ToolReply { - for_model: format!( - "The DM's full system prompt for this turn:\n\n{}", - self.prompt - ), - public: Text::from(Line::from(Span::styled(display, dim_style()))), - screened: Text::from(Line::from(Span::styled( - format!("{LABEL} ({} chars)", self.prompt.len()), - dim_style(), - ))), + for_model: format!("The DM's full system prompt for this turn:\n\n{prompt}"), + public: line.clone(), + screened: line, + cap: Visibility::Screened, }) } } diff --git a/src/dm/tools/context_tests.rs b/src/dm/tools/context_tests.rs index b7bb3fb..04c9353 100644 --- a/src/dm/tools/context_tests.rs +++ b/src/dm/tools/context_tests.rs @@ -1,23 +1,42 @@ use super::*; +use crate::dm::tools::Toolbox; +use crate::knowledge::fixtures::{self, Layer}; use serde_json::json; +/// A context tool over a stack whose whole prompt is `base`: no layers +/// and an empty mount, so nothing else joins it. +fn tool_over(base: &str) -> ContextTool { + ContextTool::new(Arc::new(ContextStack::new( + base, + &[], + Arc::new(fixtures::mount(&[])), + ))) +} + #[test] fn name_is_context() { - let tool = ContextTool::new("test prompt".to_string()); - assert_eq!(tool.name(), "context"); + assert_eq!(tool_over("test prompt").name(), "context"); } #[test] fn definition_includes_the_name() { - let tool = ContextTool::new("test prompt".to_string()); - let def = tool.definition(); + let definition = tool_over("test prompt").definition(); + + assert_eq!(definition["function"]["name"], "context"); +} + +#[test] +fn the_description_comes_from_the_sibling_markdown_file() { + let definition = tool_over("test prompt").definition(); - assert_eq!(def["function"]["name"], "context"); + let description = definition["function"]["description"].as_str().unwrap(); + assert!(description.starts_with("Reveal the DM's full system prompt")); + assert!(description.contains("renders at `screened` at most")); } #[test] fn call_returns_the_prompt_for_the_model() { - let mut tool = ContextTool::new("You are the DM.\n\nRules: actions.".to_string()); + let mut tool = tool_over("You are the DM.\n\nRules: actions."); let reply = tool.call(&json!({}), Visibility::Public).unwrap(); @@ -26,31 +45,74 @@ fn call_returns_the_prompt_for_the_model() { } #[test] -fn public_display_shows_the_full_prompt() { - let mut tool = ContextTool::new("Full prompt".to_string()); +fn call_composes_the_prompt_again_rather_than_reading_a_copy() { + let layer = Layer::empty(); + layer.write("system.md", "The first draft."); + let mut tool = ContextTool::new(Arc::new(ContextStack::new( + "Base.", + &[layer.path()], + Arc::new(fixtures::mount(&[])), + ))); + layer.write("system.md", "The second draft."); + + let reply = tool.call(&json!({}), Visibility::Public).unwrap(); + + assert!(reply.for_model.contains("The second draft.")); +} + +#[test] +fn a_layer_that_cannot_be_read_becomes_the_calls_error_text() { + let layer = Layer::empty(); + std::fs::write(layer.path().join("system.md"), [0xFF, 0xFE]).unwrap(); + let mut tool = ContextTool::new(Arc::new(ContextStack::new( + "Base.", + &[layer.path()], + Arc::new(fixtures::mount(&[])), + ))); + + let error = tool.call(&json!({}), Visibility::Public).unwrap_err(); + + assert!(error.contains("system.md")); +} + +#[test] +fn both_transcript_lines_show_the_size_and_never_the_prompt() { + let mut tool = tool_over(&"a".repeat(42)); + + let reply = tool.call(&json!({}), Visibility::Public).unwrap(); + + assert_eq!(reply.public.to_string(), "🔍 context (42 chars)"); + assert_eq!(reply.screened.to_string(), "🔍 context (42 chars)"); +} + +#[test] +fn the_reply_caps_at_screened() { + let mut tool = tool_over("secret prompt"); let reply = tool.call(&json!({}), Visibility::Public).unwrap(); - let text = reply.public.to_string(); - assert!(text.contains("🔍 context")); - assert!(text.contains("Full prompt")); + assert_eq!(reply.cap, Visibility::Screened); } #[test] -fn screened_display_shows_the_character_count() { - let mut tool = ContextTool::new("a".repeat(42)); +fn a_public_call_still_renders_only_the_screened_line() { + let mut toolbox = Toolbox::new(vec![Box::new(tool_over("secret prompt"))]); - let reply = tool.call(&json!({}), Visibility::Screened).unwrap(); + let outcome = toolbox.call("context", &json!({"visibility": "public"})); - let text = reply.screened.to_string(); - assert!(text.contains("42 chars")); + assert_eq!( + outcome.display.unwrap().to_string(), + "🔍 context (13 chars)" + ); + assert!(outcome.for_model.contains("secret prompt")); } #[test] -fn secret_call_still_returns_model_text() { - let mut tool = ContextTool::new("secret prompt".to_string()); +fn a_secret_call_renders_nothing_and_still_answers_the_model() { + let mut toolbox = Toolbox::new(vec![Box::new(tool_over("secret prompt"))]); - let reply = tool.call(&json!({}), Visibility::Secret).unwrap(); + let outcome = toolbox.call("context", &json!({"visibility": "secret"})); - assert!(reply.for_model.contains("secret prompt")); + assert_eq!(outcome.display, None); + assert!(outcome.for_model.contains("secret prompt")); } diff --git a/src/dm/tools/dice.rs b/src/dm/tools/dice.rs index e69c383..ec927e4 100644 --- a/src/dm/tools/dice.rs +++ b/src/dm/tools/dice.rs @@ -198,6 +198,7 @@ impl Tool for DiceTool { for_model: for_model(&outcome), public, screened, + cap: Visibility::Public, }) } } diff --git a/src/dm/tools/lookup.rs b/src/dm/tools/lookup.rs index 3fc910c..af1203d 100644 --- a/src/dm/tools/lookup.rs +++ b/src/dm/tools/lookup.rs @@ -14,7 +14,7 @@ use serde_json::{Value, json}; use crate::knowledge::{Mount, SearchResult, search}; -use super::{Tool, ToolReply, Visibility}; +use super::{Tool, ToolReply, Visibility, cap_for}; const LOOKUP_MD: &str = include_str!("lookup.md"); @@ -201,6 +201,19 @@ fn for_model(outcome: &Outcome, mount: &Mount) -> String { sections.join("\n\n") } +/// The strictest visibility among the entries this result names, listed +/// and inlined alike, since the transcript line is about the search as a +/// whole. +fn cap(outcome: &Outcome, mount: &Mount) -> Visibility { + cap_for( + outcome + .result + .matches + .iter() + .filter_map(|matched| mount.get(&matched.address)), + ) +} + /// One line per listed match: its address, name, and kind, when it has /// one. fn listing(result: &SearchResult) -> String { @@ -269,11 +282,16 @@ impl Tool for LookupTool { }) } + /// The model reads every match whatever its visibility says, and the + /// matches' visibility caps what the transcript may show: one + /// screened match renders the screened line even on a public call, + /// and one secret match renders nothing. fn call(&mut self, args: &Value, _visibility: Visibility) -> Result { let outcome = execute(args, &self.mount)?; let (public, screened) = render(&outcome); Ok(ToolReply { for_model: for_model(&outcome, &self.mount), + cap: cap(&outcome, &self.mount), public, screened, }) diff --git a/src/dm/tools/lookup_tests.rs b/src/dm/tools/lookup_tests.rs index 4d3f90f..d871d18 100644 --- a/src/dm/tools/lookup_tests.rs +++ b/src/dm/tools/lookup_tests.rs @@ -2,6 +2,7 @@ //! project's file-length guideline. use super::*; +use crate::dm::tools::{ToolOutcome, Toolbox}; use crate::knowledge::fixtures::{EntrySpec, mount as fixture_mount, mount_from_specs}; use crate::knowledge::{Match, SearchResult}; use serde_json::json; @@ -499,3 +500,97 @@ fn call_propagates_an_execute_error() { "`keywords` is required; give one or more keywords like `[\"fireball\"]`" ); } + +// --- The entries' own visibility --------------------------------------------- + +/// A two-entry mount: a public inn and a host whose `visibility` is +/// `host_visibility`. A search for `harbor` matches both. +fn inn_and_host(host_visibility: &str) -> Mount { + mount_from_specs(&[ + EntrySpec::new("lore/places/inn", "The Inn").with_body("A harbor tavern."), + EntrySpec::new("lore/people/host", "The Host") + .with_visibility(host_visibility) + .with_body("She runs the harbor tavern."), + ]) +} + +/// Searches for `harbor` at `visibility`, through the toolbox, so the +/// cap the tool set is applied the way a turn applies it. +fn lookup_harbor(mount: Mount, visibility: &str) -> ToolOutcome { + Toolbox::new(vec![Box::new(LookupTool::new(Arc::new(mount)))]).call( + "lookup", + &json!({ "keywords": ["harbor"], "visibility": visibility }), + ) +} + +#[test] +fn matches_with_no_visibility_field_cap_at_public() { + let mut tool = tool(goblin_mount()); + + let reply = tool + .call(&json!({ "keywords": ["goblin"] }), Visibility::Public) + .unwrap(); + + assert_eq!(reply.cap, Visibility::Public); +} + +#[test] +fn one_screened_match_caps_the_reply_at_screened() { + let mut tool = tool(inn_and_host("screened")); + + let reply = tool + .call(&json!({ "keywords": ["harbor"] }), Visibility::Public) + .unwrap(); + + assert_eq!(reply.cap, Visibility::Screened); +} + +#[test] +fn one_secret_match_caps_the_reply_at_secret() { + let mut tool = tool(inn_and_host("secret")); + + let reply = tool + .call(&json!({ "keywords": ["harbor"] }), Visibility::Public) + .unwrap(); + + assert_eq!(reply.cap, Visibility::Secret); +} + +#[test] +fn a_public_lookup_that_finds_a_screened_entry_renders_the_screened_line() { + let outcome = lookup_harbor(inn_and_host("screened"), "public"); + + assert_eq!( + outcome.display, + Some(Text::from(Line::from(Span::styled( + SCREENED_LINE, + dim_style() + )))) + ); +} + +#[test] +fn a_public_lookup_that_finds_a_secret_entry_renders_nothing() { + let outcome = lookup_harbor(inn_and_host("secret"), "public"); + + assert_eq!(outcome.display, None); +} + +#[test] +fn the_model_still_reads_every_match_in_full() { + let outcome = lookup_harbor(inn_and_host("secret"), "public"); + + assert!(outcome.for_model.contains("She runs the harbor tavern.")); +} + +#[test] +fn a_public_lookup_of_public_entries_still_names_where_it_landed() { + let outcome = lookup_harbor(inn_and_host("public"), "public"); + + assert_eq!( + outcome.display, + Some(Text::from(Line::from(Span::raw( + "📖 lookup \"harbor\" → 2 matches" + )))) + ); +} diff --git a/src/dm/tools/mark.rs b/src/dm/tools/mark.rs index fc5c90d..e12ec8a 100644 --- a/src/dm/tools/mark.rs +++ b/src/dm/tools/mark.rs @@ -195,6 +195,7 @@ impl Tool for MarkTool { for_model: for_model(&outcome, visibility), public: public_line(&outcome), screened: screened_line(&outcome), + cap: Visibility::Public, }) } } diff --git a/src/dm/tools/mod.rs b/src/dm/tools/mod.rs index 9818a3f..091aed3 100644 --- a/src/dm/tools/mod.rs +++ b/src/dm/tools/mod.rs @@ -7,10 +7,18 @@ //! concern: the `Toolbox` adds the parameter to every declaration, //! strips it from the arguments before a tool runs, and applies it to //! the tool's reply afterward. +//! +//! A reply also carries a cap, the loosest level it may render at, and +//! the `Toolbox` takes the stricter of the cap and the call. That is +//! where a knowledge entry's own `visibility` lands: a `lookup` or a +//! `read` that names a screened entry caps at `Screened`, so the model +//! cannot show the player a screened entry by asking for `public`. use ratatui::text::Text; use serde_json::{Value, json}; +use crate::knowledge::Entry; + pub mod context; pub mod dice; pub mod lookup; @@ -45,8 +53,9 @@ pub trait Tool: Send { fn call(&mut self, args: &Value, visibility: Visibility) -> Result; } -/// What a tool call produced: the text the model sees, and the two ways -/// the transcript can render it, depending on the call's visibility. +/// What a tool call produced: the text the model sees, the two ways the +/// transcript can render it, and how far the reply may be rendered at +/// all. #[derive(Debug)] pub struct ToolReply { /// The tool result message sent back to the model. @@ -56,10 +65,16 @@ pub struct ToolReply { /// The line shown in the transcript when the call is screened: the /// same event with its numbers left out. pub screened: Text<'static>, + /// The loosest level this reply may render at, whatever the call + /// asked for. `Public` leaves the call's own visibility in charge. + pub cap: Visibility, } /// How much of a tool call the player sees. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +/// +/// The variants are declared loosest first, so `Ord` compares them as +/// "how much is withheld" and `max` picks the stricter of two levels. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] pub enum Visibility { /// The player sees the full result. Public, @@ -85,6 +100,38 @@ impl Visibility { )), } } + + /// The stricter of `self` and `other`. + fn stricter(self, other: Self) -> Self { + self.max(other) + } +} + +impl From for Visibility { + fn from(visibility: crate::knowledge::Visibility) -> Self { + match visibility { + crate::knowledge::Visibility::Public => Visibility::Public, + crate::knowledge::Visibility::Screened => Visibility::Screened, + crate::knowledge::Visibility::Secret => Visibility::Secret, + } + } +} + +/// The cap a reply carries when it names `entries`: the strictest +/// visibility any one of them declares. An entry with no `visibility` +/// field is public. +/// +/// The model still gets every entry's full text, because it is the DM. +/// The cap governs the player-facing side only. +pub fn cap_for<'a>(entries: impl Iterator) -> Visibility { + entries.fold(Visibility::Public, |cap, entry| { + cap.stricter( + entry + .visibility + .map(Visibility::from) + .unwrap_or(Visibility::Public), + ) + }) } /// What dispatching one tool call produced: the tool result for the @@ -140,10 +187,11 @@ impl Toolbox { /// /// Strips `visibility` from `args` before the tool sees them, then /// applies it to the tool's reply: a public or screened call shows - /// its matching line, a secret call shows nothing. A problem at any - /// step, an unknown name, arguments that are not a JSON object, a - /// missing or invalid `visibility`, or the tool's own `Err`, becomes - /// the tool result with nothing to display. + /// its matching line, a secret call shows nothing. A reply whose cap + /// is stricter than the call was renders at the cap instead. A + /// problem at any step, an unknown name, arguments that are not a + /// JSON object, a missing or invalid `visibility`, or the tool's own + /// `Err`, becomes the tool result with nothing to display. pub fn call(&mut self, name: &str, args: &Value) -> ToolOutcome { let Some(tool) = self.tools.iter_mut().find(|tool| tool.name() == name) else { return error(unknown_tool(name, &self.tools)); @@ -215,9 +263,10 @@ fn with_visibility(mut definition: Value) -> Value { } /// Applies `visibility` to a tool's reply, picking what the transcript -/// shows. +/// shows. The reply's own cap wins whenever it is the stricter of the +/// two. fn outcome(reply: ToolReply, visibility: Visibility) -> ToolOutcome { - let display = match visibility { + let display = match visibility.stricter(reply.cap) { Visibility::Public => Some(reply.public), Visibility::Screened => Some(reply.screened), Visibility::Secret => None, diff --git a/src/dm/tools/read.rs b/src/dm/tools/read.rs index 80c0d4b..637e6ed 100644 --- a/src/dm/tools/read.rs +++ b/src/dm/tools/read.rs @@ -13,7 +13,7 @@ use serde_json::{Value, json}; use crate::knowledge::{Entry, Mount, normalize_address}; -use super::{Tool, ToolReply, Visibility}; +use super::{Tool, ToolReply, Visibility, cap_for}; const READ_MD: &str = include_str!("read.md"); @@ -111,6 +111,10 @@ impl Tool for ReadTool { }) } + /// The model reads the entry whatever its visibility says, and the + /// entry's visibility caps what the transcript may show: a screened + /// entry renders the screened line even on a public call, and a + /// secret one renders nothing. fn call(&mut self, args: &Value, _visibility: Visibility) -> Result { let entry = execute(args, &self.mount)?; let (public, screened) = render(&entry); @@ -118,6 +122,7 @@ impl Tool for ReadTool { for_model: entry.text.clone(), public, screened, + cap: cap_for(std::iter::once(&entry)), }) } } diff --git a/src/dm/tools/read_tests.rs b/src/dm/tools/read_tests.rs index 776dadf..678cc2f 100644 --- a/src/dm/tools/read_tests.rs +++ b/src/dm/tools/read_tests.rs @@ -2,7 +2,8 @@ //! project's file-length guideline. use super::*; -use crate::knowledge::fixtures::mount as fixture_mount; +use crate::dm::tools::{ToolOutcome, Toolbox}; +use crate::knowledge::fixtures::{EntrySpec, mount as fixture_mount, mount_from_specs}; use serde_json::json; /// A single-entry mount: one goblin, under `rules/monsters`. @@ -181,3 +182,103 @@ fn call_propagates_an_execute_error() { "no entry is mounted at `rules/monsters/orc`; use `lookup` to find the right address" ); } + +// --- The entry's own visibility ---------------------------------------------- + +/// A one-entry mount whose goblin declares `visibility`. +fn mount_with_visibility(visibility: &str) -> Mount { + mount_from_specs(&[EntrySpec::new("rules/monsters/goblin", "Goblin") + .with_kind("monster") + .with_visibility(visibility) + .with_body("A small, cunning creature.")]) +} + +/// Reads the goblin at `visibility`, through the toolbox, so the cap the +/// tool set is applied the way a turn applies it. +fn read_goblin(mount: Mount, visibility: &str) -> ToolOutcome { + Toolbox::new(vec![Box::new(ReadTool::new(Arc::new(mount)))]).call( + "read", + &json!({ "address": "rules/monsters/goblin", "visibility": visibility }), + ) +} + +#[test] +fn an_entry_with_no_visibility_field_caps_at_public() { + let mut tool = tool(goblin_mount()); + + let reply = tool + .call( + &json!({ "address": "rules/monsters/goblin" }), + Visibility::Public, + ) + .unwrap(); + + assert_eq!(reply.cap, Visibility::Public); +} + +#[test] +fn a_screened_entry_caps_the_reply_at_screened() { + let mut tool = tool(mount_with_visibility("screened")); + + let reply = tool + .call( + &json!({ "address": "rules/monsters/goblin" }), + Visibility::Public, + ) + .unwrap(); + + assert_eq!(reply.cap, Visibility::Screened); +} + +#[test] +fn a_secret_entry_caps_the_reply_at_secret() { + let mut tool = tool(mount_with_visibility("secret")); + + let reply = tool + .call( + &json!({ "address": "rules/monsters/goblin" }), + Visibility::Public, + ) + .unwrap(); + + assert_eq!(reply.cap, Visibility::Secret); +} + +#[test] +fn a_public_read_of_a_screened_entry_renders_the_screened_line() { + let outcome = read_goblin(mount_with_visibility("screened"), "public"); + + assert_eq!( + outcome.display, + Some(Text::from(Line::from(Span::styled( + SCREENED_LINE, + dim_style() + )))) + ); +} + +#[test] +fn a_public_read_of_a_secret_entry_renders_nothing() { + let outcome = read_goblin(mount_with_visibility("secret"), "public"); + + assert_eq!(outcome.display, None); +} + +#[test] +fn the_model_still_reads_a_secret_entry_in_full() { + let outcome = read_goblin(mount_with_visibility("secret"), "public"); + + assert!(outcome.for_model.contains("A small, cunning creature.")); +} + +#[test] +fn a_public_read_of_a_public_entry_still_names_the_address() { + let outcome = read_goblin(mount_with_visibility("public"), "public"); + + assert_eq!( + outcome.display, + Some(Text::from(Line::from(Span::raw( + "📖 rules/monsters/goblin" + )))) + ); +} diff --git a/src/dm/tools/recall.rs b/src/dm/tools/recall.rs index daa9fdc..1f65a5c 100644 --- a/src/dm/tools/recall.rs +++ b/src/dm/tools/recall.rs @@ -176,6 +176,7 @@ impl Tool for RecallTool { for_model: for_model(&outcome), public, screened, + cap: Visibility::Public, }) } } diff --git a/src/dm/tools/tools_tests.rs b/src/dm/tools/tools_tests.rs index 7efbe52..4bfffcb 100644 --- a/src/dm/tools/tools_tests.rs +++ b/src/dm/tools/tools_tests.rs @@ -1,11 +1,15 @@ use super::*; +use crate::knowledge::fixtures::{EntrySpec, mount_from_specs}; use serde_json::json; /// A tool double for exercising the toolbox: `Echo` echoes its /// arguments back as the tool result alongside canned display lines, -/// and `Failing` always returns the given error. +/// `Capped` echoes the same way behind a cap, the way a knowledge tool +/// caps a reply that names a screened entry, and `Failing` always +/// returns the given error. enum FakeTool { Echo, + Capped(Visibility), Failing(String), } @@ -31,16 +35,24 @@ impl Tool for FakeTool { fn call(&mut self, args: &Value, _visibility: Visibility) -> Result { match self { - FakeTool::Echo => Ok(ToolReply { - for_model: args.to_string(), - public: Text::raw("public line"), - screened: Text::raw("screened line"), - }), + FakeTool::Echo => Ok(echo(args, Visibility::Public)), + FakeTool::Capped(cap) => Ok(echo(args, *cap)), FakeTool::Failing(message) => Err(message.clone()), } } } +/// The reply `FakeTool` echoes: the arguments it saw, two canned lines, +/// and `cap`. +fn echo(args: &Value, cap: Visibility) -> ToolReply { + ToolReply { + for_model: args.to_string(), + public: Text::raw("public line"), + screened: Text::raw("screened line"), + cap, + } +} + /// A second minimal tool, used only to prove an unknown-name error /// lists every tool that exists, not just the first. struct OtherTool; @@ -215,6 +227,72 @@ fn a_secret_call_shows_nothing_but_still_answers_the_model() { assert_eq!(outcome.for_model, json!({"target": "goblin"}).to_string()); } +#[test] +fn a_public_call_of_a_screened_reply_shows_the_screened_line() { + let outcome = call( + FakeTool::Capped(Visibility::Screened), + json!({"target": "goblin", "visibility": "public"}), + ); + + assert_eq!(outcome.display, Some(Text::raw("screened line"))); +} + +#[test] +fn a_public_call_of_a_secret_reply_shows_nothing() { + let outcome = call( + FakeTool::Capped(Visibility::Secret), + json!({"target": "goblin", "visibility": "public"}), + ); + + assert_eq!(outcome.display, None); + assert_eq!(outcome.for_model, json!({"target": "goblin"}).to_string()); +} + +#[test] +fn a_secret_call_of_a_screened_reply_stays_secret() { + let outcome = call( + FakeTool::Capped(Visibility::Screened), + json!({"target": "goblin", "visibility": "secret"}), + ); + + assert_eq!(outcome.display, None); +} + +#[test] +fn entries_with_no_visibility_field_cap_at_public() { + let mount = mount_from_specs(&[EntrySpec::new("lore/places/inn", "The Inn")]); + + assert_eq!(cap_for(mount.entries()), Visibility::Public); +} + +#[test] +fn one_screened_entry_caps_a_whole_result_at_screened() { + let mount = mount_from_specs(&[ + EntrySpec::new("lore/places/inn", "The Inn"), + EntrySpec::new("lore/people/host", "The Host").with_visibility("screened"), + ]); + + assert_eq!(cap_for(mount.entries()), Visibility::Screened); +} + +#[test] +fn one_secret_entry_caps_a_whole_result_at_secret() { + let mount = mount_from_specs(&[ + EntrySpec::new("lore/people/host", "The Host").with_visibility("screened"), + EntrySpec::new("lore/people/assassin", "The Assassin").with_visibility("secret"), + ]); + + assert_eq!(cap_for(mount.entries()), Visibility::Secret); +} + +#[test] +fn an_explicitly_public_entry_caps_at_public() { + let mount = + mount_from_specs(&[EntrySpec::new("lore/places/inn", "The Inn").with_visibility("public")]); + + assert_eq!(cap_for(mount.entries()), Visibility::Public); +} + #[test] fn an_unknown_tool_name_lists_every_tool_that_exists() { let mut toolbox = Toolbox::new(vec![Box::new(FakeTool::Echo), Box::new(OtherTool)]); diff --git a/src/play/terminal.rs b/src/play/terminal.rs index ca360f4..298e020 100644 --- a/src/play/terminal.rs +++ b/src/play/terminal.rs @@ -24,9 +24,7 @@ use ratatui::{Terminal, TerminalOptions, Viewport}; use crate::campaign::Campaign; use crate::config::{self, Overrides}; use crate::dm::Dm; -use crate::dm::tools::{Toolbox, dice::DiceTool, lookup::LookupTool, read::ReadTool}; use crate::knowledge::Mount; -use rand::rngs::StdRng; use super::history::History; use super::keys::{self, Key, Keys}; @@ -44,16 +42,20 @@ const POLL_INTERVAL: Duration = Duration::from_millis(50); /// The terminal goes back to how it was before the game, with the /// transcript still on the screen. An init that fails part way through has /// already put the terminal in raw mode, so that path restores it too. A -/// mount that fails to open, the same as a config that fails to load, -/// ends the game before the terminal changes anything. Pinning the -/// viewport to the bottom happens after both succeed and before the -/// terminal enters raw mode, so a failure there ends the game the same -/// way. +/// mount that fails to open, a DM that fails to start, and a config that +/// fails to load all end the game before the terminal changes anything. +/// Pinning the viewport to the bottom happens after all three succeed +/// and before the terminal enters raw mode, so a failure there ends the +/// game the same way. pub fn run(overrides: &Overrides, layers: &[PathBuf], world_root: &Path) -> Result<(), String> { let config = config::load(overrides).map_err(|error| error.to_string())?; let mount = Arc::new(Mount::open(layers)?); let campaign = Campaign::open(world_root)?; let banner = super::banner(&config); + let dm = Dm::new(config, mount, layers, Some(campaign))?; + // The DM answers what `/context` shows: its own prompt and its own + // tools, so the report cannot drift from the session. + let slash_context = dm.context_report(); pin_to_bottom(VIEWPORT_HEIGHT).map_err(|error| error.to_string())?; let options = TerminalOptions { viewport: Viewport::Inline(VIEWPORT_HEIGHT), @@ -66,24 +68,9 @@ pub fn run(overrides: &Overrides, layers: &[PathBuf], world_root: &Path) -> Resu // event with the pasted text kept whole; not worth failing the game // over. let _ = execute!(std::io::stdout(), EnableBracketedPaste); - let worker = Worker::spawn(Dm::new(config, Arc::clone(&mount), layers, Some(campaign))); + let worker = Worker::spawn(dm); let history_path = world_root.join("terminal_history"); let mut history = History::load(history_path); - // Build the context prompt for the /context command. The DM's - // own context stack rebuilds it each turn; this copy is what the - // slash command shows, frozen at startup. - let context_prompt = - crate::context::ContextStack::build(include_str!("../dm/base-system.md"), layers, &mount) - .map(|ctx| ctx.prompt().to_string()) - .unwrap_or_default(); - let tool_defs = render_tool_definitions(layers); - let slash_context = format!("{context_prompt}\n\n---\n\n## Tools\n\n{tool_defs}"); - let char_count = slash_context.len(); - // Rough estimate: most tokenizers average 3.5-4 chars per token for - // English prose. 4 gives a conservative upper bound. - let est_tokens = char_count / 4; - let slash_context = - format!("{slash_context}\n\n---\n\n*{char_count} chars, ~{est_tokens} tokens*"); let played = screen::play( &mut terminal, &mut CrosstermKeys, @@ -103,56 +90,6 @@ pub fn run(overrides: &Overrides, layers: &[PathBuf], world_root: &Path) -> Resu played.map_err(|error| error.to_string()) } -/// Renders the default toolbox's definitions as a readable block for -/// the /context slash command. -fn render_tool_definitions(layers: &[PathBuf]) -> String { - let mount = match Mount::open(layers) { - Ok(m) => m, - Err(_) => return String::new(), - }; - let rng: StdRng = rand::make_rng(); - let mount_arc = Arc::new(mount); - let tools: Vec> = vec![ - Box::new(DiceTool::new(rng)), - Box::new(LookupTool::new(Arc::clone(&mount_arc))), - Box::new(ReadTool::new(mount_arc)), - ]; - let toolbox = Toolbox::new(tools); - let defs = toolbox.definitions(); - defs.iter() - .map(|def| { - let name = def["function"]["name"].as_str().unwrap_or("?"); - let desc = def["function"]["description"].as_str().unwrap_or(""); - let params = def["function"]["parameters"]["properties"] - .as_object() - .map(|props| { - let mut lines: Vec = props - .iter() - .map(|(key, val)| { - let ptype = val["type"].as_str().unwrap_or("?"); - let pdesc = val["description"].as_str().unwrap_or(""); - format!(" - `{key}` ({ptype}): {pdesc}") - }) - .collect(); - lines.sort(); - lines.join("\n") - }) - .unwrap_or_default(); - let required: Vec<&str> = def["function"]["parameters"]["required"] - .as_array() - .map(|arr| arr.iter().filter_map(|v| v.as_str()).collect()) - .unwrap_or_default(); - let required_line = if required.is_empty() { - String::new() - } else { - format!("\n required: {}", required.join(", ")) - }; - format!("### {name}\n\n{desc}\n\nParameters:\n{params}{required_line}") - }) - .collect::>() - .join("\n\n") -} - /// Puts the cursor on the line right below the inline viewport, at column /// 0, so the shell's next prompt starts on a fresh line under the /// transcript instead of wherever the terminal last left it. diff --git a/src/play/worker.rs b/src/play/worker.rs index cd8729a..6e18771 100644 --- a/src/play/worker.rs +++ b/src/play/worker.rs @@ -214,6 +214,7 @@ mod tests { &[], None, ) + .unwrap() } #[test]