diff --git a/plans/0012-the-archivist.md b/plans/0012-the-archivist.md index c9856f2..b9b9861 100644 --- a/plans/0012-the-archivist.md +++ b/plans/0012-the-archivist.md @@ -302,7 +302,10 @@ does. Where the file lives is decided with the world root in phase 10. here, and nothing more until the Archivist has run for a while. - **Which model plays the Archivist.** Per the ground rules, we pick at the last minute. It is a separate agent loop against the same - OpenAI-compatible API, and it does not have to be the DM's model. + OpenAI-compatible API, and it does not have to be the DM's model or + share the DM's reasoning setting. Each agent builds its own `Client`, + so per-agent model and reasoning choices are a config-file shape to + decide when the agent exists. - **Cadence tuning.** Turn-end, session-end, or somewhere between: the cursor makes this a knob, and we will set it by feel at the table. - **The context budget.** Richer notes make injection bigger, which diff --git a/src/chat/client.rs b/src/chat/client.rs index 33a94cc..2a4cf4d 100644 --- a/src/chat/client.rs +++ b/src/chat/client.rs @@ -63,6 +63,10 @@ pub struct Client { pub api_base: String, pub api_key: String, pub model: String, + /// How much the model deliberates before it answers, sent to the + /// provider verbatim on every request. `None` omits the field, so the + /// provider's own default applies. + pub reasoning_effort: Option, } impl Client { @@ -97,6 +101,7 @@ impl Client { messages: messages.to_vec(), stream: true, tools: tools.to_vec(), + reasoning_effort: self.reasoning_effort.clone(), }; let body = serde_json::to_vec(&request).expect("ChatRequest always serializes to JSON"); @@ -411,6 +416,7 @@ mod tests { api_base, api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, } } @@ -503,6 +509,35 @@ mod tests { assert_eq!(sent["tools"], serde_json::json!(tools)); } + #[test] + fn stream_puts_the_reasoning_effort_on_the_request_when_set() { + let (url, requests, server) = fake_server(sse_response("data: [DONE]\n\n")); + let client = Client { + reasoning_effort: Some("high".to_string()), + ..client_for(url) + }; + + client.stream(&a_message(), &[]).unwrap().for_each(drop); + + server.join().unwrap(); + let request = requests.recv().unwrap(); + let sent: serde_json::Value = serde_json::from_str(&request.body).unwrap(); + assert_eq!(sent["reasoning_effort"], "high"); + } + + #[test] + fn stream_omits_reasoning_effort_from_the_request_when_unset() { + let (url, requests, server) = fake_server(sse_response("data: [DONE]\n\n")); + let client = client_for(url); + + client.stream(&a_message(), &[]).unwrap().for_each(drop); + + server.join().unwrap(); + let request = requests.recv().unwrap(); + let sent: serde_json::Value = serde_json::from_str(&request.body).unwrap(); + assert!(sent.get("reasoning_effort").is_none()); + } + #[test] fn a_trailing_slash_on_api_base_does_not_double_the_path() { let (url, requests, server) = fake_server(sse_response("data: [DONE]\n\n")); diff --git a/src/chat/wire.rs b/src/chat/wire.rs index 87c7a80..2649a1a 100644 --- a/src/chat/wire.rs +++ b/src/chat/wire.rs @@ -8,6 +8,11 @@ pub struct ChatRequest { pub stream: bool, #[serde(skip_serializing_if = "Vec::is_empty")] pub tools: Vec, + /// How much the model deliberates before it answers, sent to the + /// provider verbatim. Omitted from the request when `None`, so the + /// provider's own default applies. + #[serde(skip_serializing_if = "Option::is_none")] + pub reasoning_effort: Option, } /// One message in a chat history. @@ -132,6 +137,7 @@ mod tests { ], stream: true, tools: vec![], + reasoning_effort: None, }; let value = serde_json::to_value(&request).unwrap(); @@ -156,6 +162,7 @@ mod tests { messages: vec![], stream: true, tools: vec![json!({"type": "function", "function": {"name": "roll"}})], + reasoning_effort: None, }; let value = serde_json::to_value(&request).unwrap(); @@ -166,6 +173,21 @@ mod tests { ); } + #[test] + fn chat_request_serializes_reasoning_effort_when_present() { + let request = ChatRequest { + model: "gpt-4o-mini".to_string(), + messages: vec![], + stream: true, + tools: vec![], + reasoning_effort: Some("high".to_string()), + }; + + let value = serde_json::to_value(&request).unwrap(); + + assert_eq!(value["reasoning_effort"], json!("high")); + } + #[test] fn assistant_role_serializes_lowercase() { let message = message(Role::Assistant, "You see a torch-lit hall."); diff --git a/src/cli.rs b/src/cli.rs index 9ef9743..95e6fed 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -70,6 +70,11 @@ enum Command { /// The model to play with. #[arg(long, value_name = "NAME")] model: Option, + /// How much the model deliberates before it answers. Valid + /// values depend on the provider, for example low, high, max on + /// DeepInfra's DeepSeek models. + #[arg(long, value_name = "LEVEL")] + reasoning_effort: Option, }, } @@ -106,8 +111,16 @@ pub fn run( }) => run_srd_verify(layer_root), Some(Command::Roll { notation }) => run_roll(¬ation), #[cfg(not(coverage))] - Some(Command::Sandbox { api_base, model }) => run_sandbox( - &crate::config::Overrides { api_base, model }, + Some(Command::Sandbox { + api_base, + model, + reasoning_effort, + }) => run_sandbox( + &crate::config::Overrides { + api_base, + model, + reasoning_effort, + }, session_layers, ), } diff --git a/src/config.rs b/src/config.rs index e9f22a7..1e4e0c1 100644 --- a/src/config.rs +++ b/src/config.rs @@ -13,7 +13,8 @@ const DEFAULT_API_KEY_ENV: &str = "STORIED_API_KEY"; /// An example config file, shown to the player when none exists yet. const EXAMPLE_CONFIG: &str = "api_base = \"https://api.deepinfra.com/v1/openai\"\n\ model = \"deepseek-ai/DeepSeek-V4-Flash\"\n\ -api_key_env = \"DEEPINFRA_API_KEY\"\n"; +api_key_env = \"DEEPINFRA_API_KEY\"\n\ +reasoning_effort = \"high\"\n"; /// Where storied talks and as what model. #[derive(Debug)] @@ -21,13 +22,18 @@ pub struct Config { pub api_base: String, pub api_key: String, pub model: String, + /// How much the model deliberates before it answers, sent to the + /// provider verbatim. `None` omits the field from every request, so + /// the provider's own default applies. + pub reasoning_effort: Option, } -/// Values that replace the config file's `api_base` and `model` when set, -/// typically from command-line flags. +/// Values that replace the config file's `api_base`, `model`, and +/// `reasoning_effort` when set, typically from command-line flags. pub struct Overrides { pub api_base: Option, pub model: Option, + pub reasoning_effort: Option, } /// The config file's fields as TOML describes them. `api_base` and `model` @@ -39,6 +45,7 @@ struct RawConfig { api_base: Option, model: Option, api_key_env: Option, + reasoning_effort: Option, } /// Everything that can go wrong loading the config file. @@ -177,10 +184,13 @@ fn resolve_config( path: path.to_path_buf(), })?; + let reasoning_effort = overrides.reasoning_effort.clone().or(raw.reasoning_effort); + Ok(Config { api_base, api_key, model, + reasoning_effort, }) } diff --git a/src/config_tests.rs b/src/config_tests.rs index 761361a..08a46ca 100644 --- a/src/config_tests.rs +++ b/src/config_tests.rs @@ -13,6 +13,7 @@ fn no_overrides() -> Overrides { Overrides { api_base: None, model: None, + reasoning_effort: None, } } @@ -145,6 +146,7 @@ fn overrides_replace_the_files_api_base_and_model() { let overrides = Overrides { api_base: Some("https://override.test".to_string()), model: Some("override-model".to_string()), + reasoning_effort: None, }; let config = resolve_config(contents, &path, &overrides, &env_with_key("sk-live")).unwrap(); @@ -153,6 +155,66 @@ fn overrides_replace_the_files_api_base_and_model() { assert_eq!(config.model, "override-model"); } +#[test] +fn reasoning_effort_is_none_when_absent_from_the_file() { + let path = PathBuf::from("/config/storied/config.toml"); + let contents = "api_base = \"https://example.test\"\n\ + model = \"test-model\"\n\ + api_key_env = \"TEST_API_KEY\"\n"; + + let config = + resolve_config(contents, &path, &no_overrides(), &env_with_key("sk-live")).unwrap(); + + assert_eq!(config.reasoning_effort, None); +} + +#[test] +fn reasoning_effort_loads_from_the_file() { + let path = PathBuf::from("/config/storied/config.toml"); + let contents = "api_base = \"https://example.test\"\n\ + model = \"test-model\"\n\ + api_key_env = \"TEST_API_KEY\"\n\ + reasoning_effort = \"high\"\n"; + + let config = + resolve_config(contents, &path, &no_overrides(), &env_with_key("sk-live")).unwrap(); + + assert_eq!(config.reasoning_effort, Some("high".to_string())); +} + +#[test] +fn the_reasoning_effort_override_replaces_the_files_value() { + let path = PathBuf::from("/config/storied/config.toml"); + let contents = "api_base = \"https://example.test\"\n\ + model = \"test-model\"\n\ + api_key_env = \"TEST_API_KEY\"\n\ + reasoning_effort = \"high\"\n"; + let overrides = Overrides { + api_base: None, + model: None, + reasoning_effort: Some("low".to_string()), + }; + + let config = resolve_config(contents, &path, &overrides, &env_with_key("sk-live")).unwrap(); + + assert_eq!(config.reasoning_effort, Some("low".to_string())); +} + +#[test] +fn the_example_config_parses_and_sets_a_reasoning_effort() { + let path = PathBuf::from("/config/storied/config.toml"); + + let config = resolve_config( + EXAMPLE_CONFIG, + &path, + &no_overrides(), + &env_with_key("sk-live"), + ) + .unwrap(); + + assert_eq!(config.reasoning_effort, Some("high".to_string())); +} + #[test] fn a_syntax_error_is_a_malformed_error() { let path = PathBuf::from("/config/storied/config.toml"); diff --git a/src/dm/dm_session_start_tests.rs b/src/dm/dm_session_start_tests.rs index 0415198..4a71b8a 100644 --- a/src/dm/dm_session_start_tests.rs +++ b/src/dm/dm_session_start_tests.rs @@ -265,6 +265,7 @@ fn a_transcript_that_cannot_be_read_fails_the_dm_at_startup() { api_base: "http://127.0.0.1:0".to_string(), api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, }, Arc::new(fixtures::mount(&[])), &[], diff --git a/src/dm/dm_tests.rs b/src/dm/dm_tests.rs index 7fae866..3d1c827 100644 --- a/src/dm/dm_tests.rs +++ b/src/dm/dm_tests.rs @@ -23,6 +23,24 @@ pub(super) fn dm_for(api_base: String) -> Dm { api_base, api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, + }, + Arc::new(fixtures::mount(&[])), + &[], + None, + ) + .unwrap() +} + +/// A `Dm` whose config sets `reasoning_effort`, so a test can assert the +/// value reaches the request. +fn dm_with_reasoning_effort(api_base: String, reasoning_effort: &str) -> Dm { + Dm::new( + Config { + api_base, + api_key: "sk-test".to_string(), + model: "gpt-4o-mini".to_string(), + reasoning_effort: Some(reasoning_effort.to_string()), }, Arc::new(fixtures::mount(&[])), &[], @@ -39,6 +57,7 @@ pub(super) fn dm_with_campaign(api_base: String, world: &TempDir) -> Dm { api_base, api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, }, Arc::new(fixtures::mount(&[])), &[], @@ -167,6 +186,19 @@ fn the_request_body_has_the_system_prompt_first_and_the_input_last() { assert_eq!(last["content"], "I open the door."); } +#[test] +fn a_turn_sends_the_configured_reasoning_effort() { + let (url, requests, server) = fake_server(vec![sse_response("data: [DONE]\n\n")]); + let mut dm = dm_with_reasoning_effort(url, "high"); + + dm.turn("I open the door.").unwrap(); + + server.join().unwrap(); + let request = requests.recv().unwrap(); + let sent: serde_json::Value = serde_json::from_str(&request.body).unwrap(); + assert_eq!(sent["reasoning_effort"], "high"); +} + #[test] fn a_turn_with_a_campaign_logs_its_narration_to_the_transcript() { let body = "data: {\"choices\":[{\"delta\":{\"content\":\"You wake.\"},\"finish_reason\":null}]}\n\n\ @@ -348,6 +380,7 @@ fn dm_over(api_base: String, layers: &[PathBuf]) -> Result { api_base, api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, }, Arc::new(fixtures::mount(&[])), layers, diff --git a/src/dm/dm_tool_round_tests.rs b/src/dm/dm_tool_round_tests.rs index 858f20c..715bbf6 100644 --- a/src/dm/dm_tool_round_tests.rs +++ b/src/dm/dm_tool_round_tests.rs @@ -31,6 +31,7 @@ fn dm_with_seeded_dice(api_base: String, seed: u64) -> Dm { api_base, api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, }, toolbox, empty_context(), @@ -194,6 +195,7 @@ fn a_post_turn_transcript_write_failure_appends_a_warning_instead_of_failing_the api_base: url, api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, }, Arc::new(fixtures::mount(&[])), &[], diff --git a/src/dm/mod.rs b/src/dm/mod.rs index 8dbf8f1..5fb871f 100644 --- a/src/dm/mod.rs +++ b/src/dm/mod.rs @@ -211,6 +211,7 @@ impl Dm { api_base: config.api_base, api_key: config.api_key, model: config.model, + reasoning_effort: config.reasoning_effort, }; let mut history = vec![system_message(&context.prompt()?)]; diff --git a/src/dm/report_tests.rs b/src/dm/report_tests.rs index e41ad2b..41c359b 100644 --- a/src/dm/report_tests.rs +++ b/src/dm/report_tests.rs @@ -35,6 +35,7 @@ fn dm(layers: &[PathBuf], world: &TempDir) -> Dm { api_base: "http://127.0.0.1:0".to_string(), api_key: "sk-test".to_string(), model: "gpt-4o-mini".to_string(), + reasoning_effort: None, }, Arc::new(fixtures::mount(&[])), layers, diff --git a/src/play/mod.rs b/src/play/mod.rs index 86a456b..afd1fdd 100644 --- a/src/play/mod.rs +++ b/src/play/mod.rs @@ -50,6 +50,7 @@ mod tests { api_base: "https://api.example.test/v1".to_string(), api_key: "sk-test".to_string(), model: "a-model".to_string(), + reasoning_effort: None, }; assert_eq!(banner(&config), "a-model @ https://api.example.test/v1"); diff --git a/src/play/worker.rs b/src/play/worker.rs index 93e6abb..8ac8b44 100644 --- a/src/play/worker.rs +++ b/src/play/worker.rs @@ -273,6 +273,7 @@ mod tests { api_base, api_key: "sk-test".to_string(), model: "a-model".to_string(), + reasoning_effort: None, }, Arc::new(fixtures::mount(&[])), &[], @@ -288,6 +289,7 @@ mod tests { api_base, api_key: "sk-test".to_string(), model: "a-model".to_string(), + reasoning_effort: None, }, Arc::new(fixtures::mount(&[])), &[],