diff --git a/core/crates/solstone-core-generate/tests/wire.rs b/core/crates/solstone-core-generate/tests/wire.rs index ae367869b..ccc12555f 100644 --- a/core/crates/solstone-core-generate/tests/wire.rs +++ b/core/crates/solstone-core-generate/tests/wire.rs @@ -375,6 +375,47 @@ fn bundled_local_invalid_provider_response_refuses() { ); } +/// A truncated **plain-text** completion is NOT a refusal, and that is worth pinning rather +/// than assuming. +/// +/// The strict validator maps a truncating finish reason to an error only when `json_output` is +/// set, so a truncated plain-text answer crosses the boundary as `generated` with its +/// `finish_reason` intact. The evidence reaches the caller; the classification does not. +/// `incomplete-text` is therefore a declared refusal reason this boundary does not emit — the +/// fan-out re-derives it for itself after the call. +/// +/// ⛔ Do not "fix" this by making it refuse. It is the deliberate behaviour of the +/// result-returning entry point, which exists so a caller decides for itself, and changing it +/// here would change what every one-shot consumer records. +#[test] +fn bundled_local_truncated_text_generates_and_carries_its_finish_reason() { + let stub = LocalStub::with_completion(Completion { + text: "a partial answ", + finish_reason: "length", + }); + let journal = Journal::bundled_local(stub.port); + let mut truncating = request(); + truncating.json_output = false; + let output = spawn_v2(&journal, &truncating); + stub.finish(); + + assert_eq!(output.status.code(), Some(0)); + let response = solstone_core_generate::decode_one_shot_response( + std::str::from_utf8(&output.stdout).unwrap(), + ) + .unwrap(); + let GenerateResponse::Generated(generated) = response else { + panic!("a truncated plain-text completion is reported as generated, not refused") + }; + // ⚠ Normalised on the way through: the endpoint said "length" and the caller sees + // "max_tokens". A consumer matching the raw provider spelling would miss the truncation + // entirely, which is exactly why this is pinned rather than assumed. + assert_eq!( + generated.finish_reason, "max_tokens", + "the caller's only signal that the answer was cut off must survive the boundary" + ); +} + #[test] fn bundled_local_non_responsive_output_refuses() { bundled_refusal( diff --git a/docs/GENERATE.md b/docs/GENERATE.md index b5d355be8..ad3e8025c 100644 --- a/docs/GENERATE.md +++ b/docs/GENERATE.md @@ -169,6 +169,20 @@ identifier, and two spelling the same concept in different cases: | `think/brain_cli.RUNTIME_REASON_CODES` (an alias import) | 41 | command-line presentation | | `think/brain_health.LOCAL_RUNTIME_REASON_CODES` | 8 | local health grouping | +⚠ **Two declared reasons are reachable only through the raising entry points, not through this +boundary**, and a caller reads the evidence instead: + +| reason | what a one-shot caller reads instead | +|---|---| +| `schema-validation-failed` | `schema_validation.valid` on the **generated** response | +| `incomplete-text` | `finish_reason` on the **generated** response | + +🔴 **So a completion the provider cut off arrives as `generated`, not as a refusal**, and it is the +caller's job to notice. That is deliberate — the result-returning path exists so a caller decides for +itself — but it means **`finish_reason` is load-bearing, not decorative**. ⚠ And it is *normalised* on +the way through: an endpoint saying `length` reaches the caller as `max_tokens`, so a consumer matching +the provider's spelling sees nothing wrong. + ⛔ **Wiring the 16 into a caller loses both decisions this contract exists to deliver**: 24 of the 43 are blocking and only 4 of those are in the 16, and the sole non-retryable code — `non_responsive` — is in the 43 and not in the 16. The fixture carries the set this contract uses, so no caller has to