返回 CodeWhale
chat.rs
根目录 / crates / tui / src / client / chat.rs
1 //! Chat Completions API helpers for DeepSeek's OpenAI-compatible endpoint.
2 //!
3 //! This is the production code path. Streaming (`create_message_stream`),
4 //! request building (`build_chat_messages*`), and SSE parsing
5 //! (`parse_sse_chunk_with_reasoning_style`) all live here.
6
7 use std::collections::HashMap;
8 use std::io::Write;
9 use std::pin::Pin;
10 use std::time::Duration;
11
12 use anyhow::{Context, Result, bail};
13 use serde::{Deserialize, Serialize};
14 use serde_json::{Value, json};
15 use tokio::time::timeout as tokio_timeout;
16
17 use crate::config::{
18 TOGETHER_INKLING_MODEL, is_exact_direct_moonshot_k3_route, is_exact_kimi_code_k3_route,
19 is_exact_zai_chat_route, is_exact_zai_forced_thinking_route, is_exact_zai_tiered_effort_route,
20 is_kimi_code_membership_model, minimax_m3_route_uses_max_completion_tokens,
21 moonshot_base_url_is_exact_kimi_code, wire_model_for_provider_route,
22 };
23
24 use crate::config::ProviderKind;
25 use crate::llm_client::StreamEventBox;
26 use crate::llm_client::sanitize_http_error_body;
27 use crate::logging;
28 use codewhale_models::{
29 ContentBlock, ContentBlockStart, Delta, Message, MessageDelta, MessageRequest, MessageResponse,
30 StreamEvent, SystemPrompt, Tool, ToolCaller, Usage, is_openai_gpt_56_api_model,
31 model_is_openai_reasoning_family, model_supports_reasoning,
32 };
33
34 use super::prepared::WireDialect;
35 use super::role_placement::{RolePlacement, role_placement};
36 use super::wire::{extract_sse_data_value, flush_sse_line, push_sse_event_data, take_sse_line};
37 use super::{
38 CodewhaleClient, ERROR_BODY_MAX_BYTES, SSE_BACKPRESSURE_HIGH_WATERMARK,
39 SSE_BACKPRESSURE_SLEEP_MS, SSE_MAX_LINES_PER_CHUNK, acquire_stream_buffer,
40 apply_reasoning_effort, bounded_error_text, from_api_tool_name, parse_usage,
41 release_stream_buffer, system_to_instructions, to_api_tool_name,
42 };
43 use codewhale_config::route::RouteLimits;
44 use codewhale_models::Role;
45
46 fn apply_provider_token_limit(
47 body: &mut Value,
48 provider: ProviderKind,
49 base_url: &str,
50 model: &str,
51 max_tokens: u32,
52 ) {
53 let use_max_completion_tokens = provider == ProviderKind::XiaomiMimo
54 || (provider == ProviderKind::Openai && model_is_openai_reasoning_family(model))
55 || minimax_m3_route_uses_max_completion_tokens(provider, base_url, model)
56 || is_exact_direct_moonshot_k3_route(provider, base_url, model);
57 if !use_max_completion_tokens {
58 return;
59 }
60
61 if let Some(object) = body.as_object_mut() {
62 object.remove("max_tokens");
63 }
64 body["max_completion_tokens"] = json!(max_tokens);
65 }
66
67 fn apply_openai_reasoning_effort(
68 body: &mut Value,
69 provider: ProviderKind,
70 model: &str,
71 effort: Option<&str>,
72 ) {
73 let model_lower = model.trim().to_ascii_lowercase();
74 let is_gpt_56 =
75 provider == ProviderKind::Openai && is_openai_gpt_56_api_model(model_lower.as_str());
76 let is_openai_reasoning =
77 provider == ProviderKind::Openai && model_is_openai_reasoning_family(model);
78 let is_muse_spark = provider == ProviderKind::Meta
79 && (model_lower == "muse-spark" || model_lower.starts_with("muse-spark-"));
80 if !is_openai_reasoning && !is_muse_spark {
81 return;
82 }
83 let Some(effort) =
84 effort.and_then(|value| openai_compatible_reasoning_effort(value, is_gpt_56, !is_gpt_56))
85 else {
86 return;
87 };
88 body["reasoning_effort"] = json!(effort);
89 }
90
91 /// xAI's first-party `reasoning_effort` ladder, driven by the bundled
92 /// catalog row for the exact model id: a row that documents an `effort`
93 /// option gets the field (`xhigh` only where the row lists it — grok-4.7 and
94 /// grok-4.6 do, grok-4.5 maps it to `high`); a row without one (grok-4.3,
95 /// grok-build) or no row at all sends nothing. Grok reasoning cannot be
96 /// disabled, so `off` is sent as the documented default `high`
97 /// (<https://docs.x.ai/docs/guides/reasoning>).
98 fn apply_xai_grok_reasoning_effort(
99 body: &mut Value,
100 provider: ProviderKind,
101 base_url: &str,
102 model: &str,
103 effort: Option<&str>,
104 ) {
105 if provider != ProviderKind::Xai
106 || !codewhale_config::provider::is_exact_xai_platform_route(
107 codewhale_config::ProviderKind::Xai,
108 base_url,
109 )
110 {
111 return;
112 }
113 let Some(effort) = effort else {
114 return;
115 };
116 let model = model.trim().to_ascii_lowercase();
117 // The bundled row with Codewhale's corrections applied (#6396): the raw
118 // seed row carries Models.dev's ladder, which lists an effort control for
119 // grok-4.3 that xAI does not document.
120 let Some(row) =
121 crate::provider_lake::bundled_catalog_offering_for_model(ProviderKind::Xai, &model)
122 else {
123 return;
124 };
125 let Some(documented) = row
126 .reasoning_options
127 .iter()
128 .find(|option| option["type"] == "effort")
129 .and_then(|option| option["values"].as_array())
130 else {
131 return;
132 };
133 let supports_xhigh = documented.iter().any(|value| value == "xhigh");
134 let wire_effort = match effort.trim().to_ascii_lowercase().as_str() {
135 "auto" | "automatic" | "" => return,
136 "off" | "disabled" | "none" | "false" | "high" => "high",
137 "minimal" | "minimum" | "low" | "light" => "low",
138 "medium" | "mid" => "medium",
139 "xhigh" | "max" | "maximum" | "highest" | "ultra" | "ultracode" => {
140 if supports_xhigh {
141 "xhigh"
142 } else {
143 "high"
144 }
145 }
146 _ => return,
147 };
148 body["reasoning_effort"] = json!(wire_effort);
149 }
150
151 fn apply_inkling_reasoning_effort(
152 body: &mut Value,
153 provider: ProviderKind,
154 model: &str,
155 effort: Option<&str>,
156 ) {
157 if provider != ProviderKind::Together
158 || !model.trim().eq_ignore_ascii_case(TOGETHER_INKLING_MODEL)
159 {
160 return;
161 }
162
163 // Inkling's official chat template accepts OpenAI's top-level
164 // `reasoning_effort` field with this exact vocabulary. It does not use
165 // Together's generic `thinking` extension or the `xhigh` wire value.
166 if let Some(object) = body.as_object_mut() {
167 object.remove("thinking");
168 }
169 let Some(effort) = effort else {
170 return;
171 };
172 let wire_effort = match effort.trim().to_ascii_lowercase().as_str() {
173 "off" | "disabled" | "none" | "false" => "none",
174 "minimal" => "minimal",
175 "low" => "low",
176 "medium" | "mid" | "" => "medium",
177 "high" => "high",
178 "max" | "xhigh" | "highest" | "ultra" | "ultracode" => "max",
179 _ => return,
180 };
181 body["reasoning_effort"] = json!(wire_effort);
182 }
183
184 /// Apply Kimi Code K3's route-specific nested thinking effort after the
185 /// generic Moonshot shaping. Other Moonshot and Kimi-compatible routes accept
186 /// only the generic enabled/disabled form, so the exact endpoint and bare
187 /// model identifier are both part of this guard.
188 fn apply_kimi_code_k3_reasoning_effort(
189 body: &mut Value,
190 provider: ProviderKind,
191 base_url: &str,
192 model: &str,
193 effort: Option<&str>,
194 ) {
195 if !is_exact_kimi_code_k3_route(provider, base_url, model) {
196 return;
197 }
198 let Some(effort) = effort else {
199 return;
200 };
201
202 let thinking = match effort.trim().to_ascii_lowercase().as_str() {
203 "off" | "none" | "disabled" | "false" | "low" | "minimum" | "minimal" | "light" => {
204 json!({ "type": "enabled", "effort": "low" })
205 }
206 "medium" | "high" => json!({ "type": "enabled", "effort": "high" }),
207 "xhigh" | "ultra" | "max" => json!({ "type": "enabled", "effort": "max" }),
208 _ => return,
209 };
210
211 // K3 uses the nested `thinking.effort` dialect. Do not leave an
212 // OpenAI-style effort value behind if another shaping layer was added
213 // before this route-specific override.
214 if let Some(object) = body.as_object_mut() {
215 object.remove("reasoning_effort");
216 }
217 body["thinking"] = thinking;
218 }
219
220 /// Apply Moonshot's direct K3 reasoning dialect.
221 ///
222 /// The pay-as-you-go K3 endpoint is always-thinking and accepts only the
223 /// top-level `reasoning_effort` values low/high/max. In particular, a generic
224 /// Moonshot `thinking: {type: disabled}` payload is not truthful for this
225 /// route. Treat a legacy raw `off` as the lowest supported tier defensively;
226 /// route-aware callers normalize it before it reaches this layer.
227 fn apply_direct_moonshot_k3_reasoning_effort(
228 body: &mut Value,
229 provider: ProviderKind,
230 base_url: &str,
231 model: &str,
232 effort: Option<&str>,
233 ) {
234 if !is_exact_direct_moonshot_k3_route(provider, base_url, model) {
235 return;
236 }
237
238 if let Some(object) = body.as_object_mut() {
239 object.remove("thinking");
240 object.remove("reasoning_effort");
241 }
242 let Some(effort) = effort else {
243 return;
244 };
245 let wire_effort = match effort.trim().to_ascii_lowercase().as_str() {
246 "off" | "none" | "disabled" | "false" | "low" | "minimum" | "minimal" | "light" => "low",
247 "medium" | "mid" | "high" | "" => "high",
248 "xhigh" | "ultra" | "max" | "highest" | "ultracode" => "max",
249 // `auto` and unknown legacy values leave the field omitted so the
250 // direct API owns its documented default (`max`).
251 _ => return,
252 };
253 body["reasoning_effort"] = json!(wire_effort);
254 }
255
256 /// Keep Z.ai controls on exact first-party routes only. The tiered-effort GLM
257 /// models (5.2, and the forced-thinking 5.3 family) receive the documented
258 /// top-level effort, GLM-5.1 and GLM-5-Turbo keep only the generic thinking
259 /// toggle, and compatible gateways receive neither field because their
260 /// request dialect is not known from provider/model selection alone.
261 fn apply_zai_route_reasoning_controls(
262 body: &mut Value,
263 provider: ProviderKind,
264 base_url: &str,
265 model: &str,
266 effort: Option<&str>,
267 ) {
268 if provider != ProviderKind::Zai {
269 return;
270 }
271
272 if let Some(object) = body.as_object_mut() {
273 object.remove("reasoning_effort");
274 if !is_exact_zai_chat_route(provider, base_url) {
275 // A compatible gateway owns its own request dialect. Provider/model
276 // selection alone is not evidence that Z.ai's `thinking` object is
277 // supported there, so fail closed instead of leaking it.
278 object.remove("thinking");
279 return;
280 }
281 }
282 if !crate::config::is_exact_known_zai_reasoning_route(provider, base_url, model) {
283 if let Some(object) = body.as_object_mut() {
284 object.remove("thinking");
285 }
286 return;
287 }
288 if !is_exact_zai_tiered_effort_route(provider, base_url, model) {
289 // Exact first-party GLM-5-Turbo and GLM-5.1 keep only the generic
290 // enabled/disabled thinking control.
291 return;
292 }
293 if is_exact_zai_forced_thinking_route(provider, base_url, model) {
294 apply_zai_forced_thinking_effort(body, effort);
295 return;
296 }
297 match effort
298 .map(|value| value.trim().to_ascii_lowercase())
299 .as_deref()
300 {
301 Some("high") => body["reasoning_effort"] = json!("high"),
302 Some("xhigh") | Some("max") | Some("highest") | Some("ultra") | Some("ultracode") => {
303 body["reasoning_effort"] = json!("max");
304 }
305 // Off, lower tiers, omitted effort, and unknown legacy values retain
306 // only the generic Z.ai thinking control.
307 _ => {}
308 }
309 }
310
311 /// GLM-5.3 and GLM-5.3-Flash are forced-thinking on the exact first-party
312 /// Z.ai route: `thinking.type: "disabled"` is rejected with an error and
313 /// `reasoning_effort` accepts only low/high/max. The generic Z.ai layer emits
314 /// `disabled` for `off`, so a request that was valid for GLM-5.2 fails on
315 /// 5.3. Rewrite that payload the way the vendor migration note prescribes —
316 /// keep thinking enabled and send the lowest tier — and map the remaining
317 /// aliases onto the three documented values, leaving unknown legacy values
318 /// omitted so the API owns its documented default (`max`).
319 fn apply_zai_forced_thinking_effort(body: &mut Value, effort: Option<&str>) {
320 let thinking_disabled = body
321 .get("thinking")
322 .and_then(|thinking| thinking.get("type"))
323 .and_then(Value::as_str)
324 == Some("disabled");
325 if thinking_disabled {
326 body["thinking"] = json!({
327 "type": "enabled",
328 "clear_thinking": false,
329 });
330 }
331 let Some(effort) = effort else {
332 return;
333 };
334 let wire_effort = match effort.trim().to_ascii_lowercase().as_str() {
335 "off" | "none" | "disabled" | "false" | "low" | "minimum" | "minimal" | "light" => "low",
336 "medium" | "mid" | "high" => "high",
337 "xhigh" | "max" | "highest" | "ultra" | "ultracode" => "max",
338 _ => return,
339 };
340 body["reasoning_effort"] = json!(wire_effort);
341 }
342
343 /// Add MiniMax's Chat-only reasoning controls only when endpoint and model
344 /// prove the exact first-party M3 route. A provider label alone is not enough
345 /// to send MiniMax-specific fields to a compatible gateway or unknown model.
346 fn apply_minimax_route_reasoning_controls(
347 body: &mut Value,
348 provider: ProviderKind,
349 base_url: &str,
350 model: &str,
351 effort: Option<&str>,
352 ) {
353 if provider != ProviderKind::Minimax {
354 return;
355 }
356 if let Some(object) = body.as_object_mut() {
357 object.remove("reasoning_split");
358 object.remove("thinking");
359 }
360 if !crate::config::is_exact_minimax_m3_route(provider, base_url, model) {
361 return;
362 }
363
364 body["reasoning_split"] = json!(true);
365 match effort
366 .map(|value| value.trim().to_ascii_lowercase())
367 .as_deref()
368 {
369 Some("off" | "disabled" | "none" | "false") => {
370 body["thinking"] = json!({ "type": "disabled" });
371 }
372 Some(
373 "low" | "minimal" | "medium" | "mid" | "high" | "xhigh" | "max" | "highest" | "ultra"
374 | "ultracode" | "",
375 ) => {
376 body["thinking"] = json!({ "type": "adaptive" });
377 }
378 _ => {}
379 }
380 }
381
382 /// Model Studio's OpenAI-compatible API uses its own top-level reasoning
383 /// controls. Keep them on verified Alibaba Chat Completions routes: a custom
384 /// `base_url` points the same provider identity at an arbitrary gateway, and
385 /// that gateway must not be handed Alibaba's dialect.
386 ///
387 /// This is the *sole* writer of Model Studio reasoning fields —
388 /// `apply_reasoning_effort` deliberately writes nothing for the `Modelstudio*`
389 /// identities — so the strip below runs for all four variants, including the
390 /// two Anthropic-dialect ones. Those normally reach the Messages adapter
391 /// instead, but `wire = "openai"` can route them here, and an unmatched
392 /// `enable_thinking` left in the body would then go out unguarded.
393 fn apply_modelstudio_route_reasoning_controls(
394 body: &mut Value,
395 provider: ProviderKind,
396 base_url: &str,
397 model: &str,
398 effort: Option<&str>,
399 ) {
400 if !matches!(
401 provider,
402 ProviderKind::ModelstudioTokenPlan
403 | ProviderKind::ModelstudioTokenPlanAnthropic
404 | ProviderKind::ModelstudioCodingPlan
405 | ProviderKind::ModelstudioCodingPlanAnthropic
406 ) {
407 return;
408 }
409
410 if let Some(object) = body.as_object_mut() {
411 object.remove("thinking");
412 object.remove("enable_thinking");
413 object.remove("preserve_thinking");
414 object.remove("reasoning_effort");
415 }
416 if !is_exact_modelstudio_chat_route(provider, base_url) {
417 return;
418 }
419
420 let thinking_only = modelstudio_model_is_thinking_only(model);
421 if !thinking_only && !modelstudio_model_is_hybrid(model) {
422 return;
423 }
424
425 let thinking_enabled = !modelstudio_effort_disables_thinking(effort);
426 // Thinking-only models emit `reasoning_content` but reject an
427 // enable/disable control. Hybrid models use `enable_thinking`.
428 if !thinking_only {
429 body["enable_thinking"] = json!(thinking_enabled);
430 }
431 if modelstudio_model_supports_preserve_thinking(model) {
432 // Model Studio otherwise drops assistant `reasoning_content` from the
433 // next turn's context. This applies even when the provider default
434 // leaves thinking enabled and no explicit UI effort was selected.
435 body["preserve_thinking"] = json!(thinking_only || thinking_enabled);
436 }
437 if !thinking_only
438 && thinking_enabled
439 && let Some(effort) = effort.and_then(modelstudio_reasoning_effort_for_model)
440 && modelstudio_model_supports_reasoning_effort(model)
441 {
442 body["reasoning_effort"] = json!(effort);
443 }
444 }
445
446 /// Fail-closed host guard: only Alibaba's own OpenAI-compatible Chat
447 /// Completions URL shapes count. Anything else (a proxy, a self-hosted
448 /// gateway, a typo) gets the Model Studio fields stripped and nothing added.
449 fn is_exact_modelstudio_chat_route(provider: ProviderKind, base_url: &str) -> bool {
450 let trimmed = base_url.trim().trim_end_matches('/').to_ascii_lowercase();
451 let Some((host, path)) = trimmed
452 .strip_prefix("https://")
453 .and_then(|rest| rest.split_once('/'))
454 else {
455 return false;
456 };
457
458 // Includes Token Plan's default and workspace-scoped
459 // `{workspace}.<region>.maas.aliyuncs.com/compatible-mode/v1` hosts.
460 let token_plan_chat = host.ends_with(".maas.aliyuncs.com") && path == "compatible-mode/v1";
461 let coding_plan_chat = host == "coding-intl.dashscope.aliyuncs.com" && path == "v1";
462 // Alibaba's classic pay-as-you-go DashScope endpoints serve the same
463 // models and the same dialect; leaving them off the allowlist silently
464 // stripped every reasoning control on a genuine Alibaba host
465 // (2026-08-04 review). The intl spelling matches the repo's own
466 // provider defaults.
467 let classic_dashscope_chat = matches!(
468 host,
469 "dashscope.aliyuncs.com" | "dashscope-intl.aliyuncs.com"
470 ) && path == "compatible-mode/v1";
471
472 match provider {
473 // The primary Model Studio provider selects Coding Plan through
474 // `mode = "coding-plan"`, which resolves this base URL without
475 // changing the provider enum. Legacy Coding Plan identities remain
476 // supported as well, so recognize either official Chat route for the
477 // complete Model Studio OpenAI family. The `*Anthropic` identities
478 // speak the Messages dialect and are never verified here.
479 ProviderKind::ModelstudioTokenPlan | ProviderKind::ModelstudioCodingPlan => {
480 token_plan_chat || coding_plan_chat || classic_dashscope_chat
481 }
482 _ => false,
483 }
484 }
485
486 fn is_exact_modelstudio_thinking_only_route(
487 provider: ProviderKind,
488 base_url: &str,
489 model: &str,
490 ) -> bool {
491 is_exact_modelstudio_chat_route(provider, base_url) && modelstudio_model_is_thinking_only(model)
492 }
493
494 fn modelstudio_effort_disables_thinking(effort: Option<&str>) -> bool {
495 effort.is_some_and(|value| {
496 matches!(
497 value.trim().to_ascii_lowercase().as_str(),
498 "off" | "disabled" | "none" | "false"
499 )
500 })
501 }
502
503 /// Models with no enable/disable control at all. `models_dev.bundled.json`
504 /// lists `qwen3.8-max` as `thinking: always_on` and gives `qwen3.8-max-preview`
505 /// effort/budget options with no `toggle`, so sending `enable_thinking` to
506 /// either is at best ignored and at worst a 400.
507 fn modelstudio_model_is_thinking_only(model: &str) -> bool {
508 let model = model.trim().to_ascii_lowercase();
509 matches!(
510 model.as_str(),
511 "qwen3.8-max"
512 | "qwen3.8-max-preview"
513 // Kimi K2.7 Code is always-thinking. Keep both Alibaba-hosted and
514 // Moonshot-supplied exact IDs separate from hybrid Kimi variants
515 // so we never send the unsupported enable_thinking switch.
516 | "kimi-k2.7-code"
517 | "kimi/kimi-k2.7-code"
518 | "kimi/kimi-k2.7-code-highspeed"
519 )
520 }
521
522 fn modelstudio_model_is_hybrid(model: &str) -> bool {
523 let model = model.trim().to_ascii_lowercase();
524 model.starts_with("qwen3.7-")
525 || model.starts_with("qwen3.6-")
526 || model.starts_with("qwen3.5-")
527 || model.starts_with("qwen3-")
528 || model.starts_with("deepseek-v4")
529 || model.starts_with("deepseek-v3.2")
530 || model.starts_with("deepseek-v3.1")
531 || model.starts_with("kimi-k2.6")
532 || matches!(model.as_str(), "kimi/kimi-k2.6")
533 || model.starts_with("kimi-k2.5")
534 || model.starts_with("glm-")
535 }
536
537 fn modelstudio_model_supports_preserve_thinking(model: &str) -> bool {
538 let model = model.trim().to_ascii_lowercase();
539 matches!(
540 model.as_str(),
541 "qwen3.7-max"
542 | "qwen3.7-max-us"
543 | "qwen3.7-max-2026-05-17"
544 | "qwen3.7-max-2026-05-20"
545 | "qwen3.7-max-2026-06-08"
546 | "qwen3.7-max-preview"
547 | "qwen3.7-plus"
548 | "qwen3.7-plus-us"
549 | "qwen3.7-plus-2026-05-26"
550 | "qwen3.6-max-preview"
551 | "qwen3.6-plus"
552 | "qwen3.6-plus-2026-04-02"
553 | "qwen3.6-flash"
554 | "qwen3.6-flash-2026-04-16"
555 | "kimi-k2.6"
556 | "kimi-k2.7-code"
557 | "kimi/kimi-k2.6"
558 | "kimi/kimi-k2.7-code"
559 | "kimi/kimi-k2.7-code-highspeed"
560 )
561 }
562
563 fn modelstudio_model_supports_reasoning_effort(model: &str) -> bool {
564 let model = model.trim().to_ascii_lowercase();
565 model.starts_with("deepseek-v4") || matches!(model.as_str(), "glm-5.2" | "glm-5.1" | "glm-5")
566 }
567
568 fn modelstudio_reasoning_effort_for_model(effort: &str) -> Option<&'static str> {
569 match effort.trim().to_ascii_lowercase().as_str() {
570 // Model Studio documents low and medium as aliases for high.
571 "minimal" | "low" | "medium" | "mid" | "high" | "" => Some("high"),
572 "xhigh" | "max" | "highest" | "ultra" | "ultracode" => Some("max"),
573 _ => None,
574 }
575 }
576
577 /// Final reasoning-control pass shared by streaming and non-streaming Chat
578 /// Completions requests. Route-specific shapers run after the generic provider
579 /// layer so they can remove fields that are invalid for their exact endpoint.
580 pub(super) fn apply_route_reasoning_controls(
581 body: &mut Value,
582 provider: ProviderKind,
583 base_url: &str,
584 model: &str,
585 effort: Option<&str>,
586 ) {
587 apply_reasoning_effort(body, effort, provider);
588 apply_modelstudio_route_reasoning_controls(body, provider, base_url, model, effort);
589 apply_minimax_route_reasoning_controls(body, provider, base_url, model, effort);
590 apply_inkling_reasoning_effort(body, provider, model, effort);
591 apply_openai_reasoning_effort(body, provider, model, effort);
592 apply_xai_grok_reasoning_effort(body, provider, base_url, model, effort);
593 apply_direct_moonshot_k3_reasoning_effort(body, provider, base_url, model, effort);
594 apply_kimi_code_k3_reasoning_effort(body, provider, base_url, model, effort);
595 apply_zai_route_reasoning_controls(body, provider, base_url, model, effort);
596 apply_mistral_route_reasoning_controls(body, provider, base_url, model, effort);
597 apply_google_reasoning_effort(body, base_url, model, effort);
598 }
599
600 /// Mistral's polymorphic reasoning-content contract is only proven on its
601 /// first-party Chat Completions endpoints. A configured `mistral` provider may
602 /// point at an arbitrary OpenAI-compatible gateway, so provider identity alone
603 /// is not enough to opt that route into Mistral's request or response dialect.
604 fn is_exact_mistral_chat_route(provider: ProviderKind, base_url: &str) -> bool {
605 if provider != ProviderKind::Mistral {
606 return false;
607 }
608 let trimmed = base_url.trim().trim_end_matches('/').to_ascii_lowercase();
609 let Some((host, path)) = trimmed
610 .strip_prefix("https://")
611 .and_then(|rest| rest.split_once('/'))
612 else {
613 return false;
614 };
615 matches!(
616 host,
617 "api.mistral.ai" | "api.eu.mistral.ai" | "api.us.mistral.ai"
618 ) && path == "v1"
619 }
620
621 /// Google's OpenAI-compatibility route, identified by the **resolved base
622 /// URL** rather than by provider identity. Thought signatures are captured
623 /// from tool-call `extra_content.google.thought_signature` and replayed on
624 /// the assistant tool-call messages of later turns; thinking models fail
625 /// closed when a replayed call has no signature.
626 ///
627 /// The endpoint carries the signature contract, not the config row that
628 /// happens to name it: a manually configured `kind="openai-compatible"`
629 /// provider ([`ProviderKind::Custom`]) pointed at this exact host and path is
630 /// byte-for-byte the same endpoint as the built-in `google` row, so it must
631 /// preserve and replay signatures the same way. The converse still holds —
632 /// a `google` row pointed at some other gateway is not this route and never
633 /// carries Google-only fields off-endpoint.
634 fn is_google_openai_compat_chat_route(base_url: &str) -> bool {
635 let trimmed = base_url.trim().trim_end_matches('/').to_ascii_lowercase();
636 let Some((host, path)) = trimmed
637 .strip_prefix("https://")
638 .and_then(|rest| rest.split_once('/'))
639 else {
640 return false;
641 };
642 host == "generativelanguage.googleapis.com" && path == "v1beta/openai"
643 }
644
645 /// Gemini models whose thinking makes thought signatures load-bearing on
646 /// the OpenAI-compat route. Gemini 2.5 Flash-Lite ships with thinking off
647 /// by default, so a missing signature there degrades with a warning
648 /// instead of failing the turn.
649 fn google_model_requires_thought_signatures(model: &str) -> bool {
650 let model = model.trim().to_ascii_lowercase();
651 // Google names the same model both ways on this endpoint, and a route
652 // configured as `models/gemini-3-pro` matched none of the prefixes below:
653 // the model that most needs a signature looked like one that needs none,
654 // so the fail-closed check waved it through and Google rejected the replay
655 // instead (#6018).
656 let model = model.strip_prefix("models/").unwrap_or(&model);
657 if model.starts_with("gemini-3") {
658 return true;
659 }
660 if model.starts_with("gemini-2.5-pro") {
661 return true;
662 }
663 model.starts_with("gemini-2.5-flash") && !model.starts_with("gemini-2.5-flash-lite")
664 }
665
666 /// Google's compatibility endpoint accepts the ordinary `reasoning_effort`
667 /// field across Gemini 2.5 and 3. A top-level `google` object is rejected;
668 /// native thinking controls would require `extra_body.google` instead.
669 /// Use one control, since the endpoint rejects overlapping effort and native
670 /// thinking settings. https://ai.google.dev/gemini-api/docs/openai#thinking
671 fn apply_google_reasoning_effort(
672 body: &mut serde_json::Value,
673 base_url: &str,
674 model: &str,
675 effort: Option<&str>,
676 ) {
677 if !is_google_openai_compat_chat_route(base_url) {
678 return;
679 }
680 let Some(effort) = effort else {
681 return;
682 };
683 let model = model.trim().to_ascii_lowercase();
684 let model = model.strip_prefix("models/").unwrap_or(&model);
685 let can_disable = model.starts_with("gemini-2.5-") && !model.starts_with("gemini-2.5-pro");
686 let effort = match effort.trim().to_ascii_lowercase().as_str() {
687 "off" | "disabled" | "none" | "false" if can_disable => "none",
688 // Gemini 3 and 2.5 Pro cannot disable thinking. The compatibility
689 // layer maps minimal to the selected model's lowest supported level.
690 "off" | "disabled" | "none" | "false" | "minimal" => "minimal",
691 "low" => "low",
692 "medium" | "mid" | "" => "medium",
693 "high" | "xhigh" | "max" | "highest" | "ultra" | "ultracode" => "high",
694 _ => return,
695 };
696 body["reasoning_effort"] = json!(effort);
697 }
698
699 /// Fail closed before transport when Google's OpenAI-compat route would
700 /// replay tool calls without the thought signatures Google's thinking models
701 /// require. The error names the model and the tool call and tells the
702 /// operator how to recover instead of letting Google reject or corrupt the
703 /// tool loop.
704 ///
705 /// Models whose thinking is off by default (Gemini 2.5 Flash-Lite) degrade
706 /// instead of failing — but never silently: the unsigned replay is reported
707 /// through the same warning path the reasoning-replay sanitizer uses, so a
708 /// later tool-turn failure has a receipt. Only tool-call identifiers and the
709 /// model id are logged; signature bytes never are.
710 fn validate_google_thought_signature_replay(
711 base_url: &str,
712 model: &str,
713 messages: &[Value],
714 ) -> Result<()> {
715 if !is_google_openai_compat_chat_route(base_url) {
716 return Ok(());
717 }
718 let requires_signatures = google_model_requires_thought_signatures(model);
719 let mut unsigned_call_ids: Vec<&str> = Vec::new();
720 for message in messages {
721 let Some(tool_calls) = message.get("tool_calls").and_then(Value::as_array) else {
722 continue;
723 };
724 for call in tool_calls {
725 let missing = call
726 .pointer("/extra_content/google/thought_signature")
727 .and_then(Value::as_str)
728 .is_none();
729 if missing {
730 let id = call.get("id").and_then(Value::as_str).unwrap_or("?");
731 if requires_signatures {
732 anyhow::bail!(
733 "Gemini model `{model}` requires a thought signature to replay tool call \
734 `{id}`, but none was captured (the turn predates signature capture, or \
735 the provider omitted it). Start a new session before using tools on \
736 this route."
737 );
738 }
739 unsigned_call_ids.push(id);
740 }
741 }
742 }
743 if !unsigned_call_ids.is_empty() {
744 // Bounded: identifiers only, and only the first few of them.
745 let sample = unsigned_call_ids
746 .iter()
747 .take(3)
748 .copied()
749 .collect::<Vec<_>>()
750 .join(", ");
751 tracing::warn!(
752 model = %model,
753 unsigned_tool_calls = unsigned_call_ids.len(),
754 sample_tool_call_ids = %sample,
755 "replaying tool calls without Google thought signatures on the Gemini \
756 OpenAI-compatible route; later signed tool turns may be rejected"
757 );
758 }
759 Ok(())
760 }
761
762 /// Captured Google signatures ride on tool calls as
763 /// `extra_content.google.thought_signature`. Only Google's OpenAI-compat
764 /// endpoint may carry them on the wire; every other route gets them stripped
765 /// so a route switch never leaks Google-only fields to a foreign gateway.
766 ///
767 /// Returns how many tool calls lost a signature, so the caller can report a
768 /// route switch that silently drops signed history instead of dropping it
769 /// without a receipt. Never returns or logs the signature bytes.
770 fn strip_google_tool_call_extra_content(messages: &mut [Value]) -> usize {
771 let mut stripped = 0usize;
772 for message in messages {
773 let Some(tool_calls) = message.get_mut("tool_calls").and_then(Value::as_array_mut) else {
774 continue;
775 };
776 for call in tool_calls {
777 if let Some(extra) = call.get_mut("extra_content")
778 && let Some(obj) = extra.as_object_mut()
779 {
780 if obj.remove("google").is_some() {
781 stripped += 1;
782 }
783 if obj.is_empty() {
784 call.as_object_mut().map(|c| c.remove("extra_content"));
785 }
786 }
787 }
788 }
789 stripped
790 }
791
792 fn mistral_model_has_adjustable_reasoning(model: &str) -> bool {
793 let model = model.trim().to_ascii_lowercase();
794 model.starts_with("mistral-medium") || model.starts_with("mistral-small")
795 }
796
797 fn mistral_model_has_native_reasoning(model: &str) -> bool {
798 model.trim().to_ascii_lowercase().starts_with("magistral")
799 }
800
801 fn mistral_model_supports_reasoning(model: &str) -> bool {
802 mistral_model_has_adjustable_reasoning(model) || mistral_model_has_native_reasoning(model)
803 }
804
805 fn mistral_reasoning_effort_wire_value(effort: &str) -> Option<&'static str> {
806 match effort.trim().to_ascii_lowercase().as_str() {
807 "off" | "disabled" | "none" | "false" => Some("none"),
808 "high" | "xhigh" | "max" | "highest" | "ultra" | "ultracode" => Some("high"),
809 _ => None,
810 }
811 }
812
813 /// Rewrite assistant messages that carry `reasoning_content` back into the
814 /// polymorphic `content: [{type: thinking, thinking: [{type: text, text: ...}],
815 /// closed: bool}, {type: text, text: ...}]` shape that Mistral la Plateforme
816 /// emits and accepts on replay. Mistral tolerates plain-string history in a
817 /// thinking-capable conversation, but replaying the original thinking trace
818 /// keeps multi-turn reasoning quality high per the official docs
819 /// (docs.mistral.ai/capabilities/reasoning). Non-assistant messages and
820 /// assistant messages without stored thinking are left untouched.
821 fn reshape_mistral_messages_for_reasoning_replay(messages: &mut [Value]) {
822 for message in messages.iter_mut() {
823 let Some(object) = message.as_object_mut() else {
824 continue;
825 };
826 if object.get("role").and_then(Value::as_str) != Some("assistant") {
827 continue;
828 }
829 let Some(reasoning) = object.remove("reasoning_content") else {
830 continue;
831 };
832 let reasoning_text = reasoning
833 .as_str()
834 .map(str::to_string)
835 .filter(|s| !s.trim().is_empty());
836 let Some(reasoning_text) = reasoning_text else {
837 continue;
838 };
839 let text_content = object
840 .get("content")
841 .and_then(Value::as_str)
842 .map(str::to_string);
843 let mut blocks = vec![json!({
844 "type": "thinking",
845 "thinking": [{"type": "text", "text": reasoning_text}],
846 "closed": true,
847 })];
848 if let Some(text) = text_content.filter(|s| !s.trim().is_empty()) {
849 blocks.push(json!({"type": "text", "text": text}));
850 }
851 object.insert("content".to_string(), Value::Array(blocks));
852 }
853 }
854
855 /// Extract thinking and text content from a Mistral polymorphic `content`
856 /// value. Mistral la Plateforme returns `content` as either a plain string
857 /// (default) or an array of typed blocks (`{type: "thinking", thinking:
858 /// [{type: "text", text: "..."}], closed: bool}` and `{type: "text", text:
859 /// "..."}`). This helper flattens the thinking sub-array into a single
860 /// string and returns any inline text separately. It ignores plain-string
861 /// `content` (returns `(None, None)`) so the shared string fallback still
862 /// runs for non-reasoning responses.
863 fn extract_mistral_polymorphic_content(value: &Value) -> (Option<String>, Option<String>) {
864 let Some(array) = value.get("content").and_then(Value::as_array) else {
865 return (None, None);
866 };
867 let mut thinking = String::new();
868 let mut text = String::new();
869 for block in array {
870 let Some(kind) = block.get("type").and_then(Value::as_str) else {
871 continue;
872 };
873 match kind {
874 "thinking" => {
875 if let Some(inner) = block.get("thinking").and_then(Value::as_array) {
876 for sub in inner {
877 if let Some(sub_text) = sub
878 .get("text")
879 .and_then(Value::as_str)
880 .filter(|s| !s.is_empty())
881 {
882 thinking.push_str(sub_text);
883 }
884 }
885 } else if let Some(inline) = block
886 .get("thinking")
887 .and_then(Value::as_str)
888 .filter(|s| !s.is_empty())
889 {
890 thinking.push_str(inline);
891 }
892 }
893 "text" => {
894 if let Some(sub_text) = block
895 .get("text")
896 .and_then(Value::as_str)
897 .filter(|s| !s.is_empty())
898 {
899 text.push_str(sub_text);
900 }
901 }
902 _ => {}
903 }
904 }
905 let thinking = (!thinking.is_empty()).then_some(thinking);
906 let text = (!text.is_empty()).then_some(text);
907 (thinking, text)
908 }
909
910 fn apply_mistral_route_reasoning_controls(
911 body: &mut Value,
912 provider: ProviderKind,
913 base_url: &str,
914 model: &str,
915 effort: Option<&str>,
916 ) {
917 if provider != ProviderKind::Mistral {
918 return;
919 }
920 if let Some(object) = body.as_object_mut() {
921 object.remove("thinking");
922 object.remove("reasoning_effort");
923 }
924 if !is_exact_mistral_chat_route(provider, base_url)
925 || !mistral_model_has_adjustable_reasoning(model)
926 {
927 return;
928 }
929 let Some(effort) = effort else {
930 return;
931 };
932 if let Some(wire) = mistral_reasoning_effort_wire_value(effort) {
933 body["reasoning_effort"] = json!(wire);
934 }
935 }
936
937 /// The direct K3 Chat Completions schema exposes fixed sampling behavior and
938 /// omits `temperature` and `top_p`. Strip legacy/generic values only from the
939 /// exact first-party route so compatible gateways keep their own contract.
940 /// Source: <https://platform.kimi.ai/docs/guide/kimi-k3-quickstart> (verified 2026-07-20).
941 fn apply_direct_moonshot_k3_fixed_sampling(
942 body: &mut Value,
943 provider: ProviderKind,
944 base_url: &str,
945 model: &str,
946 ) {
947 if !is_exact_direct_moonshot_k3_route(provider, base_url, model) {
948 return;
949 }
950 if let Some(object) = body.as_object_mut() {
951 object.remove("temperature");
952 object.remove("top_p");
953 }
954 }
955
956 /// Kimi Code's documented membership models own their sampling behavior.
957 /// Strip generic controls only on the exact first-party membership route;
958 /// custom gateways and unknown model ids retain their own wire contract.
959 /// Source: <https://www.kimi.com/code/docs/en/third-party-tools/codex.html>
960 /// (verified 2026-08-26).
961 fn apply_kimi_code_fixed_sampling(
962 body: &mut Value,
963 provider: ProviderKind,
964 base_url: &str,
965 model: &str,
966 ) {
967 if provider != ProviderKind::Moonshot
968 || !moonshot_base_url_is_exact_kimi_code(base_url)
969 || !is_kimi_code_membership_model(model)
970 {
971 return;
972 }
973 if let Some(object) = body.as_object_mut() {
974 object.remove("temperature");
975 object.remove("top_p");
976 }
977 }
978
979 fn openai_compatible_reasoning_effort(
980 effort: &str,
981 supports_max: bool,
982 supports_minimal: bool,
983 ) -> Option<&'static str> {
984 match effort.trim().to_ascii_lowercase().as_str() {
985 "off" | "disabled" | "none" | "false" => Some("none"),
986 "minimal" if supports_minimal => Some("minimal"),
987 "minimal" => Some("low"),
988 "low" => Some("low"),
989 "medium" | "mid" | "" => Some("medium"),
990 "high" => Some("high"),
991 "xhigh" => Some("xhigh"),
992 "max" | "highest" | "ultra" | "ultracode" if supports_max => Some("max"),
993 "max" | "highest" | "ultra" | "ultracode" => Some("xhigh"),
994 _ => None,
995 }
996 }
997
998 fn mirror_minimax_reasoning_details_for_messages(messages: &mut [Value]) {
999 for message in messages {
1000 if message.get("role").and_then(Value::as_str) != Some("assistant") {
1001 continue;
1002 }
1003 if message.get("reasoning_details").is_some() {
1004 continue;
1005 }
1006 let Some(reasoning) = message
1007 .get("reasoning_content")
1008 .and_then(Value::as_str)
1009 .filter(|reasoning| !reasoning.trim().is_empty())
1010 .map(str::to_string)
1011 else {
1012 continue;
1013 };
1014 message["reasoning_details"] = json!([
1015 {
1016 "type": "text",
1017 "text": reasoning,
1018 }
1019 ]);
1020 }
1021 }
1022
1023 fn mirror_minimax_reasoning_details_for_body(body: &mut Value, provider: ProviderKind) {
1024 if provider != ProviderKind::Minimax {
1025 return;
1026 }
1027 let Some(messages) = body.get_mut("messages").and_then(Value::as_array_mut) else {
1028 return;
1029 };
1030 mirror_minimax_reasoning_details_for_messages(messages);
1031 }
1032
1033 /// Sanitize every Moonshot chat tool in place, dropping only the tools whose
1034 /// parameters cannot pass MFJS compatibility validation.
1035 ///
1036 /// Per-tool degradation: a single incompatible tool (e.g. a third-party MCP
1037 /// server whose schema uses keywords outside the MFJS whitelist) is excluded
1038 /// from this request with a warning instead of failing the whole request
1039 /// before transport. The tool name is safe to log — it is already visible in
1040 /// the UI — while the error's `Display` deliberately carries no schema values.
1041 ///
1042 /// Returns the names of the dropped tools, in catalog order.
1043 fn sanitize_moonshot_chat_tools(chat_tools: &mut Vec<Value>) -> Vec<String> {
1044 let mut dropped = Vec::new();
1045 chat_tools.retain_mut(|tool| {
1046 let Some(function) = tool
1047 .as_object_mut()
1048 .and_then(|tool| tool.get_mut("function"))
1049 .and_then(Value::as_object_mut)
1050 else {
1051 return true;
1052 };
1053 let Some(parameters) = function.get_mut("parameters") else {
1054 return true;
1055 };
1056 match crate::tools::schema_sanitize::sanitize_for_kimi_parameters(parameters) {
1057 Ok(note) => {
1058 if let Some(note) = note {
1059 let description = function
1060 .get("description")
1061 .and_then(Value::as_str)
1062 .unwrap_or_default();
1063 let description = if description.is_empty() {
1064 note
1065 } else {
1066 format!("{description} {note}")
1067 };
1068 function.insert("description".to_string(), json!(description));
1069 }
1070 true
1071 }
1072 Err(error) => {
1073 let name = function
1074 .get("name")
1075 .and_then(Value::as_str)
1076 .unwrap_or("<unnamed>")
1077 .to_string();
1078 tracing::warn!(
1079 tool = %name,
1080 error = %error,
1081 "dropping Moonshot tool from this request: parameters failed safe compatibility validation"
1082 );
1083 dropped.push(name);
1084 false
1085 }
1086 }
1087 });
1088 dropped
1089 }
1090
1091 /// The final Chat Completions wire payload for one request.
1092 ///
1093 /// Produced by [`build_chat_wire_body`], the single place where a
1094 /// `MessageRequest` becomes Chat-shaped JSON. It is reached only through
1095 /// [`super::CodewhaleClient::prepare_outbound_request`], the shared outbound
1096 /// seam that the blocking transport, the streaming transport, and
1097 /// `/preview-request` all consume — so a preview cannot drift from what would
1098 /// be sent, and no other dialect is projected through this builder.
1099 ///
1100 /// Seam concept harvested from PR #1099 (`build_sanitized_chat_completion_body`)
1101 /// by TaoMu (GTC2080); re-implemented against the current client shape.
1102 pub(crate) struct ChatWireBody {
1103 /// Provider-shaped JSON body, post-sanitizers.
1104 pub(crate) body: Value,
1105 /// The model id actually placed on the wire (may differ from the
1106 /// configured/display model for routed providers).
1107 pub(crate) model: String,
1108 /// Tokens re-sent because thinking-mode replay substituted
1109 /// `reasoning_content`. Only computed on the streaming path, which is the
1110 /// only path that runs the replay sanitizer today.
1111 pub(crate) replay_input_tokens: Option<u32>,
1112 /// Wire-normalized tool names omitted because Moonshot's MFJS validator
1113 /// rejected their parameter schemas. Kept outside the wire body so the
1114 /// caller can surface one bounded diagnostic without leaking schema data.
1115 pub(crate) omitted_tool_names: Vec<String>,
1116 }
1117
1118 /// Build the Chat Completions wire body for `request`.
1119 ///
1120 /// `stream` selects the streaming shape (`stream` + `stream_options`) and, to
1121 /// preserve historical behavior exactly, also gates the thinking-mode replay
1122 /// sanitizer — the blocking path has never run it.
1123 pub(crate) fn build_chat_wire_body(
1124 request: &MessageRequest,
1125 provider: ProviderKind,
1126 base_url: &str,
1127 stream: bool,
1128 route_limits: Option<RouteLimits>,
1129 ) -> Result<ChatWireBody> {
1130 let messages = PromptBuilder::for_request(request).build_for_provider_and_route(
1131 provider,
1132 base_url,
1133 route_limits,
1134 );
1135 let model = {
1136 let wire = wire_model_for_provider_route(provider, base_url, &request.model);
1137 codewhale_models::effective_muse_wire_id(&wire).to_string()
1138 };
1139 validate_google_thought_signature_replay(base_url, &model, &messages)?;
1140 let mut body = if stream {
1141 json!({
1142 "model": model.clone(),
1143 "messages": messages,
1144 "max_tokens": request.max_tokens,
1145 "stream": true,
1146 "stream_options": {
1147 "include_usage": true
1148 },
1149 })
1150 } else {
1151 json!({
1152 "model": model.clone(),
1153 "messages": messages,
1154 "max_tokens": request.max_tokens,
1155 })
1156 };
1157 apply_provider_token_limit(&mut body, provider, base_url, &model, request.max_tokens);
1158
1159 if let Some(temperature) = request.temperature {
1160 body["temperature"] = json!(temperature);
1161 }
1162 if let Some(top_p) = request.top_p {
1163 body["top_p"] = json!(top_p);
1164 }
1165 let mut omitted_tool_names = Vec::new();
1166 if let Some(tools) = request.tools.as_ref() {
1167 let mut chat_tools: Vec<_> = tools
1168 .iter()
1169 .map(|tool| tool_to_chat_for_base_url(tool, base_url))
1170 .collect();
1171 // Moonshot function parameters must end at a plain object root.
1172 // Flatten root composition, preserve valid nested anyOf, and drop
1173 // only the tools whose parameters cannot pass MFJS validation so one
1174 // incompatible tool never sinks the whole request.
1175 if matches!(provider, crate::config::ProviderKind::Moonshot) {
1176 omitted_tool_names = sanitize_moonshot_chat_tools(&mut chat_tools);
1177 }
1178 // xAI rejects a parameters root that is not a plain object schema
1179 // (e.g. apply_patch's root `oneOf` required-groups) with a 400.
1180 if matches!(provider, crate::config::ProviderKind::Xai) {
1181 for t in &mut chat_tools {
1182 let Some(function) = t
1183 .as_object_mut()
1184 .and_then(|t| t.get_mut("function"))
1185 .and_then(|f| f.as_object_mut())
1186 else {
1187 continue;
1188 };
1189 let note = function.get_mut("parameters").and_then(|parameters| {
1190 crate::tools::schema_sanitize::sanitize_for_xai_parameters(parameters)
1191 });
1192 if let Some(note) = note
1193 && let Some(description) = function
1194 .get_mut("description")
1195 .and_then(|d| d.as_str().map(str::to_string))
1196 {
1197 function.insert(
1198 "description".to_string(),
1199 json!(format!("{description} {note}")),
1200 );
1201 }
1202 }
1203 }
1204 // When per-tool degradation (or the caller) left no tools, omit the
1205 // key entirely: an empty `tools` array — or a `tool_choice` pointing
1206 // at a dropped tool — is itself a fresh 400 on strict providers.
1207 if !chat_tools.is_empty() {
1208 body["tools"] = json!(chat_tools);
1209 }
1210 }
1211 if should_send_tool_choice_for_chat(provider, request.reasoning_effort.as_deref())
1212 && let Some(choice) = request.tool_choice.as_ref()
1213 && let Some(mapped) = map_tool_choice_for_chat(choice)
1214 {
1215 if matches!(provider, crate::config::ProviderKind::Moonshot)
1216 && let Some(name) = mapped.pointer("/function/name").and_then(Value::as_str)
1217 && omitted_tool_names.iter().any(|omitted| omitted == name)
1218 {
1219 bail!(
1220 "Moonshot cannot force tool '{name}' because its input schema is incompatible with this route"
1221 );
1222 }
1223 if body.get("tools").is_some() {
1224 body["tool_choice"] = mapped;
1225 }
1226 }
1227 apply_route_reasoning_controls(
1228 &mut body,
1229 provider,
1230 base_url,
1231 &model,
1232 request.reasoning_effort.as_deref(),
1233 );
1234 apply_direct_moonshot_k3_fixed_sampling(&mut body, provider, base_url, &model);
1235 apply_kimi_code_fixed_sampling(&mut body, provider, base_url, &model);
1236
1237 // Bulletproof final sanitizer: walk the wire payload and force
1238 // `reasoning_content` onto any assistant message that has tool_calls
1239 // but no reasoning_content. DeepSeek's thinking-mode API rejects
1240 // such messages with a 400. This is the last line of defense after
1241 // engine-side and build-side substitution; if either upstream path
1242 // misses a case (e.g. a session restored from disk, a sub-agent
1243 // adding messages directly, or a cached prefix mismatch), this pass
1244 // still produces a valid request.
1245 let replay_input_tokens = if stream {
1246 sanitize_thinking_mode_messages_for_route(
1247 &mut body,
1248 &model,
1249 request.reasoning_effort.as_deref(),
1250 provider,
1251 base_url,
1252 )
1253 } else {
1254 None
1255 };
1256 mirror_minimax_reasoning_details_for_body(&mut body, provider);
1257
1258 Ok(ChatWireBody {
1259 body,
1260 model,
1261 replay_input_tokens,
1262 omitted_tool_names,
1263 })
1264 }
1265
1266 impl CodewhaleClient {
1267 pub(super) async fn create_message_chat(
1268 &self,
1269 prepared: &super::PreparedOutboundRequest,
1270 cacheable: bool,
1271 ) -> Result<MessageResponse> {
1272 let body = &prepared.body;
1273
1274 let response_cache_key = if cacheable {
1275 let wire_body =
1276 serde_json::to_vec(&body).context("Failed to serialize Chat API cache key")?;
1277 let key = crate::llm_response_cache::ResponseCache::make_key(
1278 self.api_provider.as_str(),
1279 &self.base_url,
1280 self.path_suffix.as_deref(),
1281 &self.api_key,
1282 &wire_body,
1283 );
1284 if let Some(cached) = crate::llm_response_cache::response_cache().get(&key) {
1285 return Ok(cached);
1286 }
1287 Some(key)
1288 } else {
1289 None
1290 };
1291
1292 // The endpoint was resolved by the shared seam alongside the body, so
1293 // a route-shape decision (e.g. DeepSeek's strict-tools `/beta` path)
1294 // cannot be made twice with two different answers.
1295 let url = prepared.endpoint.url.as_str();
1296 let response = self.send_json_with_retry(url, body).await?;
1297
1298 let status = response.status();
1299 crate::client::record_provider_response(self.api_provider, status.as_u16());
1300 if !status.is_success() {
1301 let raw_error_text = bounded_error_text(response, ERROR_BODY_MAX_BYTES).await;
1302 let error_text = sanitize_http_error_body(
1303 Some(self.api_provider.provider().display_name()),
1304 status.as_u16(),
1305 &raw_error_text,
1306 );
1307 anyhow::bail!(
1308 "Failed to call {} Chat Completions API: HTTP {status}: {error_text}",
1309 self.api_provider.provider().display_name()
1310 );
1311 }
1312
1313 let response_text = response
1314 .text()
1315 .await
1316 .context("Failed to read Chat API response body")?;
1317 let value: Value =
1318 serde_json::from_str(&response_text).context("Failed to parse Chat API JSON")?;
1319 let parsed = parse_chat_message_for_route(&value, self.api_provider, &self.base_url)?;
1320 if let Some(key) = response_cache_key {
1321 crate::llm_response_cache::response_cache().put(key, parsed.clone());
1322 }
1323 Ok(parsed)
1324 }
1325 }
1326
1327 impl CodewhaleClient {
1328 async fn open_chat_stream_response(
1329 &self,
1330 url: &str,
1331 body: &Value,
1332 ) -> Result<(reqwest::Response, Duration)> {
1333 let open_req = self.stream_open_request();
1334 let idle_timeout = open_req.idle_timeout;
1335 let response = super::stream_entry::open_sse_response(&open_req, |policy| async move {
1336 match policy {
1337 // The prebuilt HTTP/1.1 twin carries the same default
1338 // headers/auth; send once, without the JSON retry loop
1339 // (matching the pre-seam H1-pin behavior).
1340 super::stream_entry::StreamHttpPolicy::Http1Only => {
1341 let client = super::stream_entry::client_for_policy(
1342 &self.http_client,
1343 self.http1_fallback_client(),
1344 policy,
1345 );
1346 Ok(client
1347 .post(url)
1348 .header(reqwest::header::CONTENT_TYPE, "application/json")
1349 .json(body)
1350 .send()
1351 .await?)
1352 }
1353 super::stream_entry::StreamHttpPolicy::DualWithH1Fallback => {
1354 // Stream open, not a JSON retry: the response body outlives
1355 // the open, so this path must not carry any total deadline
1356 // (`open_stream_json_with_retry`, not `send_json_with_retry`).
1357 self.open_stream_json_with_retry(url, body).await
1358 }
1359 }
1360 })
1361 .await?;
1362 Ok((response, idle_timeout))
1363 }
1364
1365 pub(super) async fn handle_chat_completion_stream(
1366 &self,
1367 prepared: super::PreparedOutboundRequest,
1368 ) -> Result<StreamEventBox> {
1369 // Try true SSE streaming via chat completions (widely supported).
1370 // Body and endpoint both come from the shared prepared-request seam,
1371 // so a preview or a non-stream call can never diverge from the
1372 // streamed request.
1373 let super::PreparedOutboundRequest {
1374 body,
1375 wire_model: model,
1376 replay_input_tokens,
1377 endpoint,
1378 ..
1379 } = prepared;
1380 let url = endpoint.url;
1381
1382 let (response, stream_idle_timeout) = self.open_chat_stream_response(&url, &body).await?;
1383
1384 let status = response.status();
1385 crate::client::record_provider_response(self.api_provider, status.as_u16());
1386 if !status.is_success() {
1387 let raw_error_text = bounded_error_text(response, ERROR_BODY_MAX_BYTES).await;
1388 let error_text = sanitize_http_error_body(
1389 Some(self.api_provider.provider().display_name()),
1390 status.as_u16(),
1391 &raw_error_text,
1392 );
1393 // If DeepSeek rejected for missing reasoning_content despite the
1394 // sanitizer, dump the offending indices so we can diagnose where
1395 // they came from on the next failure.
1396 if error_text.contains("reasoning_content") {
1397 log_thinking_mode_violations(&body);
1398 }
1399 anyhow::bail!("SSE stream request failed: HTTP {status}: {error_text}");
1400 }
1401
1402 let api_provider = self.api_provider;
1403 let base_url = self.base_url.clone();
1404
1405 // Capture transport-shape headers before we consume `response` into
1406 // `bytes_stream()`. They are surfaced in the decode-error log path so
1407 // we can tell HTTP/2 RST_STREAM from chunked-encoding corruption from
1408 // gzip-compressor failure when investigating #103.
1409 let response_headers = format_stream_headers(response.headers());
1410 let byte_stream = response.bytes_stream();
1411 let configured_reasoning_stream_style = self.reasoning_stream_style.clone();
1412
1413 let stream = async_stream::stream! {
1414 use futures_util::StreamExt;
1415
1416 // Emit a synthetic MessageStart
1417 yield Ok(StreamEvent::MessageStart {
1418 message: MessageResponse {
1419 id: String::new(),
1420 r#type: "message".to_string(),
1421 role: "assistant".to_string(),
1422 content: Vec::new(),
1423 model: model.clone(),
1424 stop_reason: None,
1425 stop_sequence: None,
1426 container: None,
1427 usage: Usage {
1428 input_tokens: 0,
1429 output_tokens: 0,
1430 ..Usage::default()
1431 },
1432 },
1433 });
1434
1435 let mut line_buf = String::new();
1436 let mut byte_buf = acquire_stream_buffer();
1437 let mut content_index: u32 = 0;
1438 let mut text_started = false;
1439 let mut thinking_started = false;
1440 let mut tool_indices: std::collections::HashMap<u32, u32> = std::collections::HashMap::new();
1441 let mut reasoning_detail_buffers: std::collections::HashMap<u32, String> = std::collections::HashMap::new();
1442 let mut inline_reasoning_tags = InlineReasoningTagState::default();
1443 let reasoning_stream_style = reasoning_stream_style_for_route(
1444 api_provider,
1445 &base_url,
1446 &model,
1447 configured_reasoning_stream_style.as_deref(),
1448 );
1449
1450 let mut byte_stream = std::pin::pin!(byte_stream);
1451 let idle = stream_idle_timeout;
1452
1453 // Telemetry for #103 stream-decode diagnostics: bytes received
1454 // since the start of this stream and last successful event time.
1455 // Surfaces in the error log when reqwest yields a chunk error so
1456 // we can tell HTTP/2 RST_STREAM from chunk-decode-failure from
1457 // gzip-corruption when investigating a flaky session.
1458 let stream_start = std::time::Instant::now();
1459 let mut last_event_at = std::time::Instant::now();
1460 let mut bytes_received: usize = 0;
1461 // Set when a `[DONE]` sentinel was seen, so the post-loop flush does
1462 // not re-process trailing post-DONE bytes.
1463 let mut saw_done = false;
1464 // A number of OpenAI-compatible providers omit `[DONE]` but send a
1465 // terminal `finish_reason`. Either is valid terminal proof. A raw
1466 // HTTP EOF with neither is not: treating that as MessageStop turns
1467 // a truncated provider response into a successful empty turn.
1468 let mut saw_finish_reason = false;
1469 // Once an error has been emitted, do not follow it with a synthetic
1470 // MessageStop (or a second, less-specific premature-EOF error).
1471 let mut stream_failed = false;
1472 // Set when a complete line or unterminated flush failed UTF-8.
1473 // Skip further data-frame parsing so U+FFFD cannot enter the transcript.
1474 let mut decode_failed = false;
1475
1476 let first_byte = super::stream_entry::first_byte_timeout(idle);
1477 'stream: loop {
1478 let wait = super::stream_entry::next_chunk_timeout(idle, first_byte, bytes_received);
1479 let chunk_result = match tokio_timeout(wait, byte_stream.next()).await {
1480 Ok(Some(result)) => result,
1481 Ok(None) => break, // Stream ended normally
1482 Err(_elapsed) => {
1483 stream_failed = true;
1484 yield Err(anyhow::anyhow!(super::stream_entry::body_timeout_message(
1485 wait,
1486 bytes_received,
1487 stream_start.elapsed(),
1488 last_event_at.elapsed(),
1489 api_provider.provider().display_name(),
1490 )));
1491 break;
1492 }
1493 };
1494 let chunk = match chunk_result {
1495 Ok(bytes) => bytes,
1496 Err(e) => {
1497 stream_failed = true;
1498 // Walk the error source chain so reqwest's underlying
1499 // hyper / h2 / io error is visible — without this the
1500 // outer "error decoding response body" message tells
1501 // us nothing about WHY the stream died.
1502 let mut error_chain = format!("{e}");
1503 let mut current: Option<&(dyn std::error::Error + 'static)> =
1504 std::error::Error::source(&e);
1505 while let Some(source) = current {
1506 error_chain.push_str(&format!(" -> {source}"));
1507 current = std::error::Error::source(source);
1508 }
1509 crate::logging::warn(format!(
1510 "Stream read error: {error_chain} \
1511 (elapsed: {}ms, bytes_received: {}, ms_since_last_event: {}, headers: {})",
1512 stream_start.elapsed().as_millis(),
1513 bytes_received,
1514 last_event_at.elapsed().as_millis(),
1515 response_headers,
1516 ));
1517 yield Err(anyhow::anyhow!("Stream read error: {e}"));
1518 break;
1519 }
1520 };
1521
1522 bytes_received = bytes_received.saturating_add(chunk.len());
1523 last_event_at = std::time::Instant::now();
1524 byte_buf.extend_from_slice(&chunk);
1525
1526 // Guard against unbounded buffer growth (e.g., malformed stream without newlines)
1527 const MAX_SSE_BUF: usize = 10 * 1024 * 1024; // 10 MB
1528 if byte_buf.len() > MAX_SSE_BUF {
1529 stream_failed = true;
1530 yield Err(anyhow::anyhow!("SSE buffer exceeded {MAX_SSE_BUF} bytes — aborting stream"));
1531 break;
1532 }
1533
1534 if byte_buf.len() > SSE_BACKPRESSURE_HIGH_WATERMARK {
1535 tokio::time::sleep(Duration::from_millis(SSE_BACKPRESSURE_SLEEP_MS)).await;
1536 }
1537
1538 // Process complete SSE lines from the buffer. Decode only after
1539 // a `\n` so an HTTP/2 DATA split mid-character cannot become
1540 // U+FFFD; genuine invalid bytes fail closed.
1541 let mut lines_processed = 0usize;
1542 loop {
1543 let line = match take_sse_line(&mut byte_buf) {
1544 Ok(Some(line)) => line,
1545 Ok(None) => break,
1546 Err(err) => {
1547 decode_failed = true;
1548 stream_failed = true;
1549 yield Err(anyhow::anyhow!("{err}"));
1550 break 'stream;
1551 }
1552 };
1553
1554 if line.is_empty() {
1555 // Empty line = event boundary, process accumulated data
1556 if !line_buf.is_empty() {
1557 let data = std::mem::take(&mut line_buf);
1558 match parse_sse_data_frame(
1559 &data,
1560 &mut content_index,
1561 &mut text_started,
1562 &mut thinking_started,
1563 &mut tool_indices,
1564 &mut reasoning_detail_buffers,
1565 &mut inline_reasoning_tags,
1566 reasoning_stream_style,
1567 ) {
1568 SseDataFrame::Done => {
1569 saw_done = true;
1570 break 'stream;
1571 }
1572 SseDataFrame::Events(events) => {
1573 for mut event in events {
1574 saw_finish_reason |= matches!(
1575 &event,
1576 StreamEvent::MessageDelta { delta, .. }
1577 if delta.stop_reason.as_deref().is_some_and(|reason| !reason.trim().is_empty())
1578 );
1579 // Stamp the client-side replay-token estimate
1580 // onto the final usage so the UI can surface
1581 // it (#30). We compute it pre-request and
1582 // overlay it on the server-reported usage at
1583 // stream completion.
1584 if let Some(tokens) = replay_input_tokens
1585 && let StreamEvent::MessageDelta {
1586 usage: Some(usage),
1587 ..
1588 } = &mut event
1589 {
1590 usage.reasoning_replay_tokens = Some(tokens);
1591 }
1592 yield Ok(event);
1593 }
1594 }
1595 }
1596 }
1597 continue;
1598 }
1599
1600 if line.starts_with(':') {
1601 // SSE comment (`: keep-alive`, `: OPENROUTER PROCESSING`).
1602 // Surface it as a ping so the engine counts a provider
1603 // that is alive but queued/thinking as progress
1604 // (#6184) instead of timing out on a live stream.
1605 yield Ok(StreamEvent::Ping);
1606 continue;
1607 }
1608
1609 if let Some(data) = extract_sse_data_value(&line)
1610 && let Err(err) = push_sse_event_data(&mut line_buf, data)
1611 {
1612 decode_failed = true;
1613 stream_failed = true;
1614 yield Err(anyhow::anyhow!("{err}"));
1615 break 'stream;
1616 }
1617 // Ignore other SSE fields (event:, id:, retry:)
1618
1619 lines_processed = lines_processed.saturating_add(1);
1620 if lines_processed >= SSE_MAX_LINES_PER_CHUNK {
1621 // Backpressure relief: hand the executor a turn so a
1622 // slow consumer is not starved. Keep draining after
1623 // that — leaving complete lines buffered would strand
1624 // them, because the outer loop only resumes draining
1625 // once ANOTHER chunk arrives and the end-of-stream
1626 // flush treats the whole remainder as a single
1627 // unterminated line.
1628 lines_processed = 0;
1629 tokio::task::yield_now().await;
1630 }
1631 }
1632 }
1633
1634 // Flush a final SSE frame that arrived without a terminating blank
1635 // line (the stream closed straight after the last `data:` line, or
1636 // that line lacked a trailing newline). Without this the final delta
1637 // — last tokens, finish_reason, and usage — is silently dropped.
1638 // Skipped after `[DONE]`, whose frame was already processed, and
1639 // after a fail-closed UTF-8 error.
1640 if !saw_done && !decode_failed {
1641 match flush_sse_line(&mut byte_buf) {
1642 Ok(Some(line)) => {
1643 if let Some(data) = extract_sse_data_value(&line)
1644 && let Err(err) = push_sse_event_data(&mut line_buf, data)
1645 {
1646 decode_failed = true;
1647 stream_failed = true;
1648 yield Err(anyhow::anyhow!("{err}"));
1649 }
1650 }
1651 Ok(None) => {}
1652 Err(err) => {
1653 decode_failed = true;
1654 stream_failed = true;
1655 yield Err(anyhow::anyhow!("{err}"));
1656 }
1657 }
1658 if !decode_failed && !line_buf.is_empty() {
1659 let data = std::mem::take(&mut line_buf);
1660 match parse_sse_data_frame(
1661 &data,
1662 &mut content_index,
1663 &mut text_started,
1664 &mut thinking_started,
1665 &mut tool_indices,
1666 &mut reasoning_detail_buffers,
1667 &mut inline_reasoning_tags,
1668 reasoning_stream_style,
1669 ) {
1670 SseDataFrame::Done => saw_done = true,
1671 SseDataFrame::Events(events) => {
1672 for mut event in events {
1673 saw_finish_reason |= matches!(
1674 &event,
1675 StreamEvent::MessageDelta { delta, .. }
1676 if delta.stop_reason.as_deref().is_some_and(|reason| !reason.trim().is_empty())
1677 );
1678 if let Some(tokens) = replay_input_tokens
1679 && let StreamEvent::MessageDelta {
1680 usage: Some(usage), ..
1681 } = &mut event
1682 {
1683 usage.reasoning_replay_tokens = Some(tokens);
1684 }
1685 yield Ok(event);
1686 }
1687 }
1688 }
1689 }
1690 }
1691
1692 // Close any open blocks — content_index points to the
1693 // currently active open block (it is only incremented
1694 // *after* a block is closed, not when opened).
1695 if thinking_started || text_started {
1696 yield Ok(StreamEvent::ContentBlockStop { index: content_index });
1697 }
1698
1699 release_stream_buffer(byte_buf);
1700 if !stream_failed && (saw_done || saw_finish_reason) {
1701 yield Ok(StreamEvent::MessageStop);
1702 } else if !stream_failed {
1703 yield Err(anyhow::anyhow!(
1704 "Chat Completions stream closed before [DONE] or finish_reason"
1705 ));
1706 }
1707 };
1708
1709 Ok(Pin::from(Box::new(stream)
1710 as Box<
1711 dyn futures_util::Stream<Item = Result<StreamEvent>> + Send,
1712 >))
1713 }
1714 }
1715
1716 // === Chat Completions Helpers ===
1717
1718 #[cfg(test)]
1719 pub(super) fn build_chat_messages(
1720 system: Option<&SystemPrompt>,
1721 messages: &[Message],
1722 model: &str,
1723 ) -> Vec<Value> {
1724 build_chat_messages_with_reasoning(
1725 system,
1726 messages,
1727 tool_result_sent_char_budget(model),
1728 should_replay_reasoning_content(model, None),
1729 false,
1730 )
1731 }
1732
1733 #[cfg(test)]
1734 pub(super) fn build_chat_messages_for_request(request: &MessageRequest) -> Vec<Value> {
1735 PromptBuilder::for_request(request).build()
1736 }
1737
1738 #[cfg(test)]
1739 pub(super) fn build_chat_messages_for_request_and_provider(
1740 request: &MessageRequest,
1741 provider: ProviderKind,
1742 ) -> Vec<Value> {
1743 build_chat_messages_for_request_and_provider_and_route(request, provider, "")
1744 }
1745
1746 /// Build a wire prompt for one fully resolved provider route.
1747 ///
1748 /// Most provider behavior is keyed only by the provider kind and model. Kimi
1749 /// Code K3 is deliberately narrower: the bare `k3` model owns reasoning
1750 /// replay only on its official membership-plan endpoint, so callers that have
1751 /// a concrete base URL must retain it through prompt construction.
1752 #[cfg(test)]
1753 pub(super) fn build_chat_messages_for_request_and_provider_and_route(
1754 request: &MessageRequest,
1755 provider: ProviderKind,
1756 base_url: &str,
1757 ) -> Vec<Value> {
1758 PromptBuilder::for_request(request).build_for_provider_and_route(provider, base_url, None)
1759 }
1760
1761 pub(crate) fn inspect_prompt_for_request(request: &MessageRequest) -> PromptInspection {
1762 PromptBuilder::for_request(request).inspect()
1763 }
1764
1765 pub(crate) fn build_cache_warmup_request(request: &MessageRequest) -> MessageRequest {
1766 PromptBuilder::for_request(request).build_cache_warmup_request()
1767 }
1768
1769 struct PromptBuilder<'a> {
1770 system: Option<&'a SystemPrompt>,
1771 messages: &'a [Message],
1772 tools: Option<&'a [Tool]>,
1773 model: &'a str,
1774 reasoning_effort: Option<&'a str>,
1775 }
1776
1777 impl<'a> PromptBuilder<'a> {
1778 fn for_request(request: &'a MessageRequest) -> Self {
1779 Self {
1780 system: request.system.as_ref(),
1781 messages: &request.messages,
1782 tools: request.tools.as_deref(),
1783 model: &request.model,
1784 reasoning_effort: request.reasoning_effort.as_deref(),
1785 }
1786 }
1787
1788 #[cfg(test)]
1789 fn build(self) -> Vec<Value> {
1790 build_chat_messages_with_reasoning(
1791 self.system,
1792 self.messages,
1793 tool_result_sent_char_budget(self.model),
1794 should_replay_reasoning_content(self.model, self.reasoning_effort),
1795 false,
1796 )
1797 }
1798
1799 fn build_for_provider_and_route(
1800 self,
1801 provider: ProviderKind,
1802 base_url: &str,
1803 route_limits: Option<RouteLimits>,
1804 ) -> Vec<Value> {
1805 let mut messages = build_chat_messages_with_reasoning(
1806 self.system,
1807 self.messages,
1808 crate::route_budget::route_inline_char_budget_for_route(
1809 provider,
1810 self.model,
1811 route_limits,
1812 ),
1813 should_replay_reasoning_content_for_provider_on_route(
1814 provider,
1815 base_url,
1816 self.model,
1817 self.reasoning_effort,
1818 ),
1819 false,
1820 );
1821 dump_system_prompt_if_requested(&messages);
1822 if provider == ProviderKind::Arcee {
1823 apply_arcee_waf_safe_message_encoding(&mut messages);
1824 }
1825 if provider == ProviderKind::Minimax {
1826 mirror_minimax_reasoning_details_for_messages(&mut messages);
1827 }
1828 if is_exact_mistral_chat_route(provider, base_url) {
1829 reshape_mistral_messages_for_reasoning_replay(&mut messages);
1830 }
1831 if !is_google_openai_compat_chat_route(base_url) {
1832 // A signature captured on Google's endpoint is meaningless — and
1833 // potentially a leak — anywhere else, so it is stripped. Say so:
1834 // the model will behave differently on the replayed tool history,
1835 // and a silent strip is exactly what made this defect invisible.
1836 let stripped = strip_google_tool_call_extra_content(&mut messages);
1837 if stripped > 0 {
1838 tracing::warn!(
1839 provider = ?provider,
1840 stripped_tool_calls = stripped,
1841 "dropping captured Google thought signatures: this route is not Google's \
1842 OpenAI-compatible endpoint, so the replayed tool history is unsigned"
1843 );
1844 }
1845 }
1846 messages
1847 }
1848
1849 fn inspect(self) -> PromptInspection {
1850 let messages = build_chat_messages_with_reasoning(
1851 self.system,
1852 self.messages,
1853 tool_result_sent_char_budget(self.model),
1854 should_replay_reasoning_content(self.model, self.reasoning_effort),
1855 true,
1856 );
1857 inspect_wire_request(self.tools, &messages)
1858 }
1859
1860 fn build_cache_warmup_request(self) -> MessageRequest {
1861 let system = stable_system_prompt(self.system);
1862 let mut messages = stable_history_messages(self.messages);
1863 let tools = self
1864 .tools
1865 .filter(|tools| !tools.is_empty())
1866 .map(<[Tool]>::to_vec);
1867 let tool_choice = tools.as_ref().map(|_| json!("none"));
1868 messages.push(Message {
1869 role: Role::User,
1870 content: vec![ContentBlock::Text {
1871 text: CACHE_WARMUP_USER_TAIL.to_string(),
1872 cache_control: None,
1873 }],
1874 });
1875
1876 MessageRequest {
1877 model: self.model.to_string(),
1878 messages,
1879 max_tokens: CACHE_WARMUP_MAX_TOKENS,
1880 system,
1881 tools,
1882 tool_choice,
1883 metadata: None,
1884 thinking: None,
1885 // Warmup has an intentionally tiny answer contract ("OK"). Do not
1886 // let hidden reasoning consume that allowance before the cacheable
1887 // prefix is accepted by the provider.
1888 reasoning_effort: Some("off".to_string()),
1889 stream: None,
1890 temperature: None,
1891 top_p: None,
1892 }
1893 }
1894 }
1895
1896 const SYSTEM_PROMPT_DUMP_ENV: &str = "CODEWHALE_DUMP_SYSTEM_PROMPT";
1897 const SYSTEM_PROMPT_DUMP_BEGIN: &str = "<<<CODEWHALE_SYSTEM_PROMPT_BEGIN>>>";
1898 const SYSTEM_PROMPT_DUMP_END: &str = "<<<CODEWHALE_SYSTEM_PROMPT_END>>>";
1899 const ARCEE_WAF_TEXT_SPLIT_TRIGGERS: &[(&str, &str, &str)] = &[("python -c", "python ", "-c")];
1900
1901 fn dump_system_prompt_if_requested(messages: &[Value]) {
1902 let Ok(flag) = std::env::var(SYSTEM_PROMPT_DUMP_ENV) else {
1903 return;
1904 };
1905 if !matches!(flag.trim(), "1" | "true" | "TRUE" | "yes" | "YES") {
1906 return;
1907 }
1908 let Some(prompt) = messages.iter().find_map(system_message_text) else {
1909 return;
1910 };
1911 let mut stderr = std::io::stderr().lock();
1912 let _ = writeln!(stderr, "{SYSTEM_PROMPT_DUMP_BEGIN}");
1913 let _ = writeln!(stderr, "{prompt}");
1914 let _ = writeln!(stderr, "{SYSTEM_PROMPT_DUMP_END}");
1915 }
1916
1917 fn system_message_text(message: &Value) -> Option<String> {
1918 if message.get("role").and_then(Value::as_str) != Some("system") {
1919 return None;
1920 }
1921 match message.get("content")? {
1922 Value::String(text) => Some(text.clone()),
1923 Value::Array(parts) => {
1924 let text = parts
1925 .iter()
1926 .filter_map(|part| part.get("text").and_then(Value::as_str))
1927 .collect::<Vec<_>>()
1928 .join("");
1929 (!text.is_empty()).then_some(text)
1930 }
1931 _ => None,
1932 }
1933 }
1934
1935 fn apply_arcee_waf_safe_message_encoding(messages: &mut [Value]) {
1936 for message in messages {
1937 if message.get("role").and_then(Value::as_str) != Some("system") {
1938 continue;
1939 }
1940 let Some(content) = message.get("content").and_then(Value::as_str) else {
1941 continue;
1942 };
1943 let Some(parts) = arcee_waf_safe_text_parts(content) else {
1944 continue;
1945 };
1946 message["content"] = json!(parts);
1947 }
1948 }
1949
1950 fn arcee_waf_safe_text_parts(content: &str) -> Option<Vec<Value>> {
1951 let mut parts = Vec::new();
1952 let mut cursor = 0usize;
1953 let mut split_any = false;
1954
1955 while cursor < content.len() {
1956 let Some((trigger_start, trigger, left, right)) = next_arcee_waf_trigger(content, cursor)
1957 else {
1958 push_text_part(&mut parts, &content[cursor..]);
1959 break;
1960 };
1961
1962 push_text_part(&mut parts, &content[cursor..trigger_start]);
1963 push_text_part(&mut parts, left);
1964 push_text_part(&mut parts, right);
1965 cursor = trigger_start + trigger.len();
1966 split_any = true;
1967 }
1968
1969 split_any.then_some(parts)
1970 }
1971
1972 fn next_arcee_waf_trigger(content: &str, cursor: usize) -> Option<(usize, &str, &str, &str)> {
1973 ARCEE_WAF_TEXT_SPLIT_TRIGGERS
1974 .iter()
1975 .filter_map(|(trigger, left, right)| {
1976 content[cursor..]
1977 .find(trigger)
1978 .map(|offset| (cursor + offset, *trigger, *left, *right))
1979 })
1980 .min_by_key(|(start, _, _, _)| *start)
1981 }
1982
1983 fn push_text_part(parts: &mut Vec<Value>, text: &str) {
1984 if !text.is_empty() {
1985 parts.push(json!({
1986 "type": "text",
1987 "text": text,
1988 }));
1989 }
1990 }
1991
1992 pub(crate) const CACHE_WARMUP_USER_TAIL: &str = "请只回复 OK";
1993 pub(crate) const CACHE_WARMUP_MAX_TOKENS: u32 = 8;
1994 /// Wire backstop for tool results (#6508). The engine already fits every
1995 /// result it gives the model to the route's inline budget, and marks any cut
1996 /// with a recovery footer. This pass only catches history that never went
1997 /// through the engine (legacy or restored raw results). Model-only inspection
1998 /// uses the catalog window; outbound requests pass their resolved route budget.
1999 /// Results with an engine recovery footer remain intact.
2000 fn tool_result_sent_char_budget(model: &str) -> usize {
2001 crate::route_budget::route_inline_char_budget(codewhale_models::context_window_for_model(model))
2002 }
2003 /// Characters of an excerpted wire result spent on its labelled header.
2004 const TOOL_RESULT_EXCERPT_FRAME_CHARS: usize = 1_024;
2005 /// Tool results shorter than this stay inline even when repeated. The
2006 /// extra prompt bytes are cheaper than adding an earlier-message reference
2007 /// for tiny command outputs.
2008 const TOOL_RESULT_DEDUP_MIN_CHARS: usize = 1_024;
2009
2010 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2011 pub(crate) struct PromptInspection {
2012 pub base_static_prefix_hash: String,
2013 pub full_request_prefix_hash: String,
2014 /// Hash of the rendered tool catalog JSON, or empty when no tools were supplied.
2015 pub tool_catalog_hash: String,
2016 pub layers: Vec<PromptLayerInspection>,
2017 }
2018
2019 /// Identifies the stable prefix that a cache warmup primes.
2020 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2021 pub(crate) struct CacheWarmupKey {
2022 pub provider: String,
2023 pub model: String,
2024 pub base_url: String,
2025 pub static_prefix_hash: String,
2026 pub tool_catalog_hash: String,
2027 pub project_pack_hash: String,
2028 pub skills_hash: String,
2029 }
2030
2031 impl CacheWarmupKey {
2032 pub(crate) fn from_inspection(
2033 provider: &str,
2034 model: &str,
2035 base_url: &str,
2036 inspection: &PromptInspection,
2037 ) -> Self {
2038 Self {
2039 provider: provider.to_string(),
2040 model: model.to_string(),
2041 base_url: base_url.to_string(),
2042 static_prefix_hash: inspection.base_static_prefix_hash.clone(),
2043 tool_catalog_hash: inspection.tool_catalog_hash.clone(),
2044 project_pack_hash: layer_hash(inspection, "Project context pack"),
2045 skills_hash: layer_hash(inspection, "Skills"),
2046 }
2047 }
2048
2049 pub(crate) fn hash_short(&self) -> String {
2050 let json = serde_json::to_string(self).unwrap_or_default();
2051 let hash = sha256_hex(json.as_bytes());
2052 hash[..hash.len().min(12)].to_string()
2053 }
2054 }
2055
2056 fn layer_hash(inspection: &PromptInspection, name: &str) -> String {
2057 inspection
2058 .layers
2059 .iter()
2060 .find(|layer| layer.name == name)
2061 .map(|layer| layer.sha256.clone())
2062 .unwrap_or_default()
2063 }
2064
2065 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2066 pub(crate) struct PromptLayerInspection {
2067 pub name: String,
2068 pub stability: PromptLayerStability,
2069 pub char_len: usize,
2070 pub byte_len: usize,
2071 /// Rough token estimate for quick before/after cache-hit reports.
2072 pub token_estimate: usize,
2073 pub sha256: String,
2074 pub tool_result: Option<ToolResultInspection>,
2075 pub turn_meta: Option<TurnMetaInspection>,
2076 }
2077
2078 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2079 pub(crate) struct ToolResultInspection {
2080 pub original_chars: usize,
2081 pub sent_chars: usize,
2082 pub truncated: bool,
2083 pub deduplicated: bool,
2084 }
2085
2086 #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2087 pub(crate) struct TurnMetaInspection {
2088 pub original_chars: usize,
2089 pub sent_chars: usize,
2090 pub deduplicated: bool,
2091 pub sha256: String,
2092 }
2093
2094 #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
2095 pub(crate) enum PromptLayerStability {
2096 Static,
2097 History,
2098 Dynamic,
2099 }
2100
2101 #[cfg(test)]
2102 impl PromptLayerStability {
2103 pub(crate) fn label(self) -> &'static str {
2104 match self {
2105 Self::Static => "static",
2106 Self::History => "history",
2107 Self::Dynamic => "dynamic",
2108 }
2109 }
2110 }
2111
2112 fn inspect_wire_request(tools: Option<&[Tool]>, messages: &[Value]) -> PromptInspection {
2113 let mut layers = Vec::new();
2114 let mut base_static_prefix_parts = Vec::new();
2115 let mut full_request_prefix_parts = Vec::new();
2116 let mut tool_catalog_hash = String::new();
2117 let mut start_index = 0;
2118
2119 if let Some(message) = messages.first() {
2120 let role = message
2121 .get("role")
2122 .and_then(Value::as_str)
2123 .unwrap_or("unknown");
2124 let content = message_content_for_inspect(message);
2125 if role == "system" {
2126 for (name, stability, body) in split_system_layers(&content) {
2127 if stability == PromptLayerStability::Static {
2128 base_static_prefix_parts.push(body.to_string());
2129 }
2130 if stability != PromptLayerStability::Dynamic {
2131 full_request_prefix_parts.push(body.to_string());
2132 }
2133 layers.push(prompt_layer(name, stability, body));
2134 }
2135 start_index = 1;
2136 }
2137 }
2138
2139 if let Some(tool_catalog) = tool_catalog_for_inspect(tools) {
2140 tool_catalog_hash = sha256_hex(tool_catalog.as_bytes());
2141 base_static_prefix_parts.push(tool_catalog.clone());
2142 full_request_prefix_parts.push(tool_catalog.clone());
2143 layers.push(prompt_layer(
2144 "Tool catalog".to_string(),
2145 PromptLayerStability::Static,
2146 &tool_catalog,
2147 ));
2148 }
2149
2150 for (index, message) in messages.iter().enumerate().skip(start_index) {
2151 let role = message
2152 .get("role")
2153 .and_then(Value::as_str)
2154 .unwrap_or("unknown");
2155 let content = message_content_for_inspect(message);
2156 let is_last = index + 1 == messages.len();
2157 let stability = if (is_last && role == "user") || role == "tool" {
2158 PromptLayerStability::Dynamic
2159 } else {
2160 PromptLayerStability::History
2161 };
2162 let name = if is_last && role == "user" {
2163 "User task".to_string()
2164 } else {
2165 format!("Message #{index} {role}")
2166 };
2167 if stability != PromptLayerStability::Dynamic {
2168 full_request_prefix_parts.push(content.clone());
2169 }
2170 let mut layer = prompt_layer(name, stability, &content);
2171 layer.tool_result = tool_result_inspection_for_message(message);
2172 layer.turn_meta = turn_meta_inspection_for_message(message);
2173 layers.push(layer);
2174 }
2175
2176 let base_static_prefix = base_static_prefix_parts.join("\n");
2177 let full_request_prefix = full_request_prefix_parts.join("\n");
2178
2179 PromptInspection {
2180 base_static_prefix_hash: sha256_hex(base_static_prefix.as_bytes()),
2181 full_request_prefix_hash: sha256_hex(full_request_prefix.as_bytes()),
2182 tool_catalog_hash,
2183 layers,
2184 }
2185 }
2186
2187 fn tool_catalog_for_inspect(tools: Option<&[Tool]>) -> Option<String> {
2188 let tools = tools.filter(|tools| !tools.is_empty())?;
2189 serde_json::to_string(&tools.iter().map(tool_to_chat).collect::<Vec<_>>()).ok()
2190 }
2191
2192 fn message_content_for_inspect(message: &Value) -> String {
2193 let mut parts = Vec::new();
2194 if let Some(content) = message.get("content").and_then(Value::as_str)
2195 && !content.is_empty()
2196 {
2197 parts.push(content.to_string());
2198 }
2199 if let Some(content) = message.get("content").and_then(Value::as_array) {
2200 for part in content {
2201 match part.get("type").and_then(Value::as_str) {
2202 Some("text") => {
2203 if let Some(text) = part.get("text").and_then(Value::as_str)
2204 && !text.is_empty()
2205 {
2206 parts.push(text.to_string());
2207 }
2208 }
2209 Some("image_url") => {
2210 let url = part
2211 .get("image_url")
2212 .and_then(|image_url| image_url.get("url"))
2213 .and_then(Value::as_str)
2214 .unwrap_or("");
2215 parts.push(format!(
2216 "[image_url:{}]",
2217 summarize_image_url_for_inspect(url)
2218 ));
2219 }
2220 _ => {}
2221 }
2222 }
2223 }
2224 if let Some(reasoning) = message.get("reasoning_content").and_then(Value::as_str)
2225 && !reasoning.is_empty()
2226 {
2227 parts.push(reasoning.to_string());
2228 }
2229 if let Some(tool_calls) = message.get("tool_calls") {
2230 parts.push(tool_calls.to_string());
2231 }
2232 parts.join("\n")
2233 }
2234
2235 fn summarize_image_url_for_inspect(url: &str) -> String {
2236 let Some((prefix, encoded)) = url.split_once(";base64,") else {
2237 return first_chars(url, 96);
2238 };
2239 format!("{prefix};base64,<{} chars>", encoded.len())
2240 }
2241
2242 fn tool_result_inspection_for_message(message: &Value) -> Option<ToolResultInspection> {
2243 if message.get("role").and_then(Value::as_str) != Some("tool") {
2244 return None;
2245 }
2246 let budget = message.get("_tool_result_budget")?;
2247 Some(ToolResultInspection {
2248 original_chars: budget
2249 .get("original_chars")
2250 .and_then(Value::as_u64)
2251 .and_then(|n| usize::try_from(n).ok())?,
2252 sent_chars: budget
2253 .get("sent_chars")
2254 .and_then(Value::as_u64)
2255 .and_then(|n| usize::try_from(n).ok())?,
2256 truncated: budget
2257 .get("truncated")
2258 .and_then(Value::as_bool)
2259 .unwrap_or(false),
2260 deduplicated: budget
2261 .get("deduplicated")
2262 .and_then(Value::as_bool)
2263 .unwrap_or(false),
2264 })
2265 }
2266
2267 fn turn_meta_inspection_for_message(message: &Value) -> Option<TurnMetaInspection> {
2268 let budget = message.get("_turn_meta_budget")?;
2269 Some(TurnMetaInspection {
2270 original_chars: budget
2271 .get("original_chars")
2272 .and_then(Value::as_u64)
2273 .and_then(|n| usize::try_from(n).ok())?,
2274 sent_chars: budget
2275 .get("sent_chars")
2276 .and_then(Value::as_u64)
2277 .and_then(|n| usize::try_from(n).ok())?,
2278 deduplicated: budget
2279 .get("deduplicated")
2280 .and_then(Value::as_bool)
2281 .unwrap_or(false),
2282 sha256: budget
2283 .get("sha256")
2284 .and_then(Value::as_str)
2285 .map(str::to_string)?,
2286 })
2287 }
2288
2289 fn split_system_layers(content: &str) -> Vec<(String, PromptLayerStability, &str)> {
2290 let markers = [
2291 ("Project context", "<project_instructions"),
2292 ("Project context pack", "## Project Context Pack"),
2293 ("Environment", "## Environment"),
2294 ("Configured instructions", "<instructions "),
2295 ("User memory", "## User Memory"),
2296 ("Current session goal", "## Current Session Goal"),
2297 ("Skills", "## Skills"),
2298 ("Core execution", "## Core Execution"),
2299 ("Compact template", "## Compact"),
2300 ("Previous session relay", "## Previous Session Relay"),
2301 ];
2302
2303 let mut starts: Vec<(usize, &str)> = markers
2304 .iter()
2305 .filter_map(|(name, marker)| content.find(marker).map(|idx| (idx, *name)))
2306 .collect();
2307 starts.sort_by_key(|(idx, _)| *idx);
2308
2309 let mut layers = Vec::new();
2310 let first_marker = starts.first().map_or(content.len(), |(idx, _)| *idx);
2311 if first_marker > 0 {
2312 layers.push((
2313 "Global system prefix".to_string(),
2314 PromptLayerStability::Static,
2315 content[..first_marker].trim(),
2316 ));
2317 }
2318
2319 for (i, (start, name)) in starts.iter().enumerate() {
2320 let end = starts.get(i + 1).map_or(content.len(), |(idx, _)| *idx);
2321 let stability = if *name == "Previous session relay" {
2322 PromptLayerStability::Dynamic
2323 } else if is_static_base_layer(name) {
2324 PromptLayerStability::Static
2325 } else {
2326 PromptLayerStability::History
2327 };
2328 layers.push(((*name).to_string(), stability, content[*start..end].trim()));
2329 }
2330
2331 if layers.is_empty() {
2332 layers.push((
2333 "Global system prefix".to_string(),
2334 PromptLayerStability::Static,
2335 content.trim(),
2336 ));
2337 }
2338 layers
2339 }
2340
2341 fn is_static_base_layer(name: &str) -> bool {
2342 matches!(
2343 name,
2344 "Global system prefix"
2345 | "Environment"
2346 | "Skills"
2347 | "Project context"
2348 | "Project context pack"
2349 | "Core execution"
2350 | "Compact template"
2351 )
2352 }
2353
2354 fn stable_system_prompt(system: Option<&SystemPrompt>) -> Option<SystemPrompt> {
2355 let instructions = system_to_instructions(system.cloned())?;
2356 let stable = split_system_layers(&instructions)
2357 .into_iter()
2358 .filter_map(|(_, stability, body)| {
2359 (stability == PromptLayerStability::Static).then_some(body)
2360 })
2361 .collect::<Vec<_>>()
2362 .join("\n\n");
2363 if stable.trim().is_empty() {
2364 None
2365 } else {
2366 Some(SystemPrompt::Text(stable))
2367 }
2368 }
2369
2370 fn stable_history_messages(messages: &[Message]) -> Vec<Message> {
2371 let mut end = messages.len();
2372 if messages
2373 .last()
2374 .is_some_and(|message| message.role.as_str() == "user")
2375 {
2376 end = end.saturating_sub(1);
2377 }
2378 messages[..end].to_vec()
2379 }
2380
2381 fn prompt_layer(
2382 name: String,
2383 stability: PromptLayerStability,
2384 content: &str,
2385 ) -> PromptLayerInspection {
2386 let char_len = content.chars().count();
2387 let token_estimate = if char_len == 0 {
2388 0
2389 } else if content.is_ascii() {
2390 (char_len / 4).max(1)
2391 } else {
2392 char_len.max(1)
2393 };
2394 PromptLayerInspection {
2395 name,
2396 stability,
2397 char_len,
2398 byte_len: content.len(),
2399 token_estimate,
2400 sha256: sha256_hex(content.as_bytes()),
2401 tool_result: None,
2402 turn_meta: None,
2403 }
2404 }
2405
2406 fn sha256_hex(bytes: &[u8]) -> String {
2407 crate::hashing::sha256_hex(bytes)
2408 }
2409
2410 #[derive(Clone)]
2411 struct PendingToolCallInfo {
2412 tool_name: String,
2413 input: Value,
2414 }
2415
2416 struct SeenToolResult {
2417 message_label: String,
2418 original_chars: usize,
2419 }
2420
2421 struct WireToolResult {
2422 content: String,
2423 original_chars: usize,
2424 sent_chars: usize,
2425 truncated: bool,
2426 deduplicated: bool,
2427 }
2428
2429 #[derive(Clone)]
2430 struct TurnMetaBudget {
2431 original_chars: usize,
2432 sent_chars: usize,
2433 deduplicated: bool,
2434 sha256: String,
2435 }
2436
2437 struct LastFullTurnMeta {
2438 sha256: String,
2439 }
2440
2441 fn render_turn_meta_for_wire(
2442 text: &str,
2443 last_full_turn_meta: &mut Option<LastFullTurnMeta>,
2444 ) -> (String, TurnMetaBudget) {
2445 let original_chars = text.chars().count();
2446 let sha = sha256_hex(text.as_bytes());
2447
2448 if last_full_turn_meta
2449 .as_ref()
2450 .is_some_and(|previous| previous.sha256 == sha)
2451 {
2452 // Keep the repeated metadata slot short without surfacing an
2453 // opaque hash the model cannot resolve.
2454 let rendered = "<turn_meta_unchanged />".to_string();
2455 let budget = TurnMetaBudget {
2456 original_chars,
2457 sent_chars: rendered.chars().count(),
2458 deduplicated: true,
2459 sha256: sha,
2460 };
2461 return (rendered, budget);
2462 }
2463
2464 *last_full_turn_meta = Some(LastFullTurnMeta {
2465 sha256: sha.clone(),
2466 });
2467 (
2468 text.to_string(),
2469 TurnMetaBudget {
2470 original_chars,
2471 sent_chars: original_chars,
2472 deduplicated: false,
2473 sha256: sha,
2474 },
2475 )
2476 }
2477
2478 fn is_turn_meta_text(text: &str) -> bool {
2479 text.trim_start().starts_with("<turn_meta>")
2480 }
2481
2482 fn turn_meta_budget_json(turn_meta: &TurnMetaBudget) -> Value {
2483 json!({
2484 "original_chars": turn_meta.original_chars,
2485 "sent_chars": turn_meta.sent_chars,
2486 "deduplicated": turn_meta.deduplicated,
2487 "sha256": turn_meta.sha256,
2488 })
2489 }
2490
2491 /// Mutating/write tools whose result body is a *confirmation* (it embeds
2492 /// the unified diff + summary of what was just written), not retrievable
2493 /// reference data. Two identical large `write_file` calls must each keep
2494 /// their full confirmation inline: collapsing the later one to a
2495 /// `<TOOL_RESULT_REF sha="..." />` makes the model lose the write-success
2496 /// context and behave as if the file is missing (issue #1695). Read-style
2497 /// tools (`read_file`, `grep_files`, `exec_shell`, …) may deduplicate medium
2498 /// outputs by pointing at an earlier full message in the same request. They
2499 /// never advertise a process-wide SHA as retrievable: that store cannot prove
2500 /// session ownership.
2501 fn is_mutation_tool(tool_name: &str) -> bool {
2502 matches!(
2503 tool_name,
2504 "write" | "edit" | "write_file" | "edit_file" | "apply_patch"
2505 )
2506 }
2507
2508 fn compact_tool_result_for_wire(
2509 tool_name: &str,
2510 input: &Value,
2511 content: &str,
2512 message_label: &str,
2513 sent_budget: usize,
2514 seen_tool_results: &mut HashMap<String, SeenToolResult>,
2515 ) -> WireToolResult {
2516 let original_chars = content.chars().count();
2517 let sha = sha256_hex(content.as_bytes());
2518
2519 // Only medium, non-mutation results can point back to a full earlier
2520 // message in this one request. Oversized results are already excerpts, so
2521 // a back-reference would falsely imply the exact bytes remain available.
2522 let dedup_eligible = (TOOL_RESULT_DEDUP_MIN_CHARS..=sent_budget).contains(&original_chars)
2523 && !is_mutation_tool(tool_name);
2524
2525 if dedup_eligible && let Some(previous) = seen_tool_results.get(&sha) {
2526 let content = format!(
2527 "<TOOL_RESULT_REF sha=\"{sha}\" original_message=\"{label}\" chars=\"{chars}\">\n\
2528 source: full content appears in {label} earlier in this request\n\
2529 </TOOL_RESULT_REF>",
2530 label = previous.message_label,
2531 chars = previous.original_chars,
2532 );
2533 return WireToolResult {
2534 sent_chars: content.chars().count(),
2535 content,
2536 original_chars,
2537 truncated: false,
2538 deduplicated: true,
2539 };
2540 }
2541
2542 if dedup_eligible {
2543 seen_tool_results.insert(
2544 sha.clone(),
2545 SeenToolResult {
2546 message_label: message_label.to_string(),
2547 original_chars,
2548 },
2549 );
2550 }
2551
2552 if original_chars <= sent_budget {
2553 return WireToolResult {
2554 content: content.to_string(),
2555 original_chars,
2556 sent_chars: original_chars,
2557 truncated: false,
2558 deduplicated: false,
2559 };
2560 }
2561
2562 // Content already bounded by the adaptive evidence envelope carries its
2563 // own honest footer: the omitted count, the on-disk artifact path, and a
2564 // recovery instruction. Truncating it again here would destroy that
2565 // recovery contract and falsely report that no session-owned artifact
2566 // was recorded, so pass it through untouched.
2567 if content.contains(crate::tools::truncate::SPILLOVER_RECOVERY_HINT) {
2568 return WireToolResult {
2569 content: content.to_string(),
2570 original_chars,
2571 sent_chars: original_chars,
2572 truncated: false,
2573 deduplicated: false,
2574 };
2575 }
2576
2577 let excerpt_chars = sent_budget.saturating_sub(TOOL_RESULT_EXCERPT_FRAME_CHARS);
2578 let head_chars = excerpt_chars * 2 / 3;
2579 let head = first_chars(content, head_chars);
2580 let tail = last_chars(content, excerpt_chars - head_chars);
2581 let kept = head.chars().count() + tail.chars().count();
2582 let omitted = original_chars.saturating_sub(kept);
2583 let compacted = format!(
2584 "[TOOL_RESULT_TRUNCATED]\n\
2585 tool_name: {tool_name}\n\
2586 command_or_query: {}\n\
2587 exit_status: {}\n\
2588 original_chars: {original_chars}\n\
2589 sha256: {sha}\n\
2590 exact_detail: unavailable; no session-owned artifact was recorded\n\
2591 first_chars:\n\
2592 {head}\n\n\
2593 [... truncated {omitted} chars from middle ...]\n\n\
2594 last_chars:\n\
2595 {tail}",
2596 tool_command_or_query(input),
2597 tool_exit_status(content)
2598 );
2599
2600 WireToolResult {
2601 sent_chars: compacted.chars().count(),
2602 content: compacted,
2603 original_chars,
2604 truncated: true,
2605 deduplicated: false,
2606 }
2607 }
2608
2609 fn tool_command_or_query(input: &Value) -> String {
2610 for key in ["command", "cmd", "query", "q", "pattern", "path", "url"] {
2611 if let Some(value) = input.get(key) {
2612 return summarize_for_metadata(value, 500);
2613 }
2614 }
2615 summarize_for_metadata(input, 500)
2616 }
2617
2618 fn tool_exit_status(content: &str) -> String {
2619 if let Ok(value) = serde_json::from_str::<Value>(content) {
2620 for key in ["exit_code", "exit_status", "status", "code"] {
2621 if let Some(value) = value.get(key) {
2622 return summarize_for_metadata(value, 120);
2623 }
2624 }
2625 }
2626
2627 for line in content.lines().take(20) {
2628 let trimmed = line.trim();
2629 for prefix in ["Exit code:", "exit code:", "Exit status:", "exit status:"] {
2630 if let Some(value) = trimmed.strip_prefix(prefix) {
2631 return value.trim().to_string();
2632 }
2633 }
2634 }
2635 "unknown".to_string()
2636 }
2637
2638 fn summarize_for_metadata(value: &Value, max_chars: usize) -> String {
2639 let raw = value
2640 .as_str()
2641 .map(str::to_string)
2642 .unwrap_or_else(|| value.to_string());
2643 let mut summarized = first_chars(&raw.replace('\n', "\\n"), max_chars);
2644 if raw.chars().count() > max_chars {
2645 summarized.push_str("...");
2646 }
2647 summarized
2648 }
2649
2650 fn first_chars(value: &str, count: usize) -> String {
2651 value.chars().take(count).collect()
2652 }
2653
2654 fn last_chars(value: &str, count: usize) -> String {
2655 let mut chars: Vec<char> = value.chars().rev().take(count).collect();
2656 chars.reverse();
2657 chars.into_iter().collect()
2658 }
2659
2660 fn merge_adjacent_user_content(previous: Value, current: Value) -> Value {
2661 match (previous, current) {
2662 (Value::String(left), Value::String(right)) => json!(format!("{left}\n\n{right}")),
2663 (left, right) => {
2664 let mut parts = Vec::new();
2665 for content in [left, right] {
2666 match content {
2667 Value::Array(items) => parts.extend(items),
2668 Value::String(text) => parts.push(json!({"type": "text", "text": text})),
2669 other => parts.push(other),
2670 }
2671 }
2672 Value::Array(parts)
2673 }
2674 }
2675 }
2676
2677 fn build_chat_messages_with_reasoning(
2678 system: Option<&SystemPrompt>,
2679 messages: &[Message],
2680 tool_result_budget: usize,
2681 include_reasoning: bool,
2682 include_tool_budget_metadata: bool,
2683 ) -> Vec<Value> {
2684 let mut out = Vec::new();
2685 let mut pending_tool_calls: HashMap<String, PendingToolCallInfo> = HashMap::new();
2686 // Chat Completions requires every result for one assistant tool-call batch
2687 // to be contiguous. Keep tool-result media aside until the complete batch
2688 // has been emitted as `role: tool` messages.
2689 let mut deferred_tool_result_images = Vec::new();
2690 let mut seen_tool_results: HashMap<String, SeenToolResult> = HashMap::new();
2691 let mut last_full_turn_meta: Option<LastFullTurnMeta> = None;
2692
2693 if let Some(instructions) = system_to_instructions(system.cloned())
2694 && !instructions.trim().is_empty()
2695 {
2696 out.push(json!({
2697 "role": "system",
2698 "content": instructions,
2699 }));
2700 }
2701
2702 // Persisted compaction keeps its summary after the bounded last round.
2703 // On strict paired chat templates a user message after a tool result is
2704 // invalid. Reorder only a generated summary immediately after a tool
2705 // result; its independent provenance block rules out quoted user text.
2706 // The session log retains every original message and tool ID.
2707 // Limitation: this normalization applies to Chat Completions only.
2708 let summary_index = messages
2709 .iter()
2710 .enumerate()
2711 .rev()
2712 .find_map(|(index, message)| {
2713 (index > 0
2714 && crate::compaction::is_wire_compaction_checkpoint_message(message)
2715 && messages[index - 1]
2716 .content
2717 .iter()
2718 .any(|block| matches!(block, ContentBlock::ToolResult { .. })))
2719 .then_some(index)
2720 });
2721 let summary_target = summary_index.and_then(|summary_index| {
2722 messages[..summary_index]
2723 .iter()
2724 .rposition(|message| {
2725 crate::runtime_handoff::classify_user_turn_prompt(message)
2726 != crate::runtime_handoff::UserTurnPromptKind::NotPrompt
2727 })
2728 .or_else(|| {
2729 messages[..summary_index]
2730 .iter()
2731 .position(|message| message.role.is_assistant_like())
2732 })
2733 });
2734 let wire_messages = (0..messages.len())
2735 .filter(|index| Some(*index) != summary_index || summary_target.is_none())
2736 .flat_map(|index| {
2737 if Some(index) == summary_target {
2738 [summary_index, Some(index)]
2739 .into_iter()
2740 .flatten()
2741 .collect::<Vec<_>>()
2742 } else {
2743 vec![index]
2744 }
2745 });
2746
2747 for message_index in wire_messages {
2748 let message = &messages[message_index];
2749 // Which wire channel this message belongs in is decided by the shared
2750 // placement table, not by an `if` chain local to this adapter.
2751 let placement = role_placement(&message.role, WireDialect::ChatCompletions);
2752 let mut text_parts = Vec::new();
2753 let mut image_parts = Vec::new();
2754 let mut thinking_parts = Vec::new();
2755 let mut tool_calls = Vec::new();
2756 let mut tool_call_infos = Vec::new();
2757 let mut tool_results: Vec<(String, String, String, Vec<Value>)> = Vec::new();
2758 let mut turn_meta_budget: Option<TurnMetaBudget> = None;
2759
2760 for block in &message.content {
2761 match block {
2762 ContentBlock::Text { text, .. } => {
2763 if is_turn_meta_text(text) {
2764 let (rendered, budget) =
2765 render_turn_meta_for_wire(text, &mut last_full_turn_meta);
2766 text_parts.push(rendered);
2767 turn_meta_budget = Some(budget);
2768 } else {
2769 text_parts.push(text.clone());
2770 }
2771 }
2772 ContentBlock::ImageUrl { image_url } => {
2773 image_parts.push(json!({
2774 "type": "image_url",
2775 "image_url": {
2776 "url": image_url.url.clone(),
2777 },
2778 }));
2779 }
2780 ContentBlock::Thinking { thinking, .. } => thinking_parts.push(thinking.clone()),
2781 ContentBlock::ToolUse {
2782 id,
2783 name,
2784 input,
2785 caller,
2786 thought_signature,
2787 ..
2788 } => {
2789 let args = serde_json::to_string(input).unwrap_or_else(|_| input.to_string());
2790 let mut call = json!({
2791 "id": id,
2792 "type": "function",
2793 "function": {
2794 "name": to_api_tool_name(name),
2795 "arguments": args,
2796 }
2797 });
2798 if let Some(signature) = thought_signature {
2799 call["extra_content"]["google"]["thought_signature"] = json!(signature);
2800 }
2801 if let Some(caller) = caller {
2802 call["caller"] = json!({
2803 "type": caller.caller_type,
2804 "tool_id": caller.tool_id,
2805 });
2806 }
2807 tool_calls.push(call);
2808 tool_call_infos.push((
2809 id.clone(),
2810 PendingToolCallInfo {
2811 tool_name: name.clone(),
2812 input: input.clone(),
2813 },
2814 ));
2815 }
2816 ContentBlock::ToolResult {
2817 tool_use_id,
2818 content,
2819 content_blocks,
2820 ..
2821 } => {
2822 let message_label = format!("Message #{message_index}");
2823 tool_results.push((
2824 tool_use_id.clone(),
2825 content.clone(),
2826 message_label,
2827 content_blocks.clone().unwrap_or_default(),
2828 ));
2829 }
2830 ContentBlock::ServerToolUse { .. }
2831 | ContentBlock::ToolSearchToolResult { .. }
2832 | ContentBlock::CodeExecutionToolResult { .. } => {}
2833 }
2834 }
2835
2836 let out_len_before_role_projection = out.len();
2837 if placement.is_assistant_channel() {
2838 let content = if placement == RolePlacement::InterruptedAssistant {
2839 format!(
2840 "{}{}",
2841 codewhale_models::INTERRUPTED_ASSISTANT_CONTEXT_PREFIX,
2842 text_parts.join("\n")
2843 )
2844 } else {
2845 text_parts.join("\n")
2846 };
2847 let mut reasoning_content = thinking_parts.join("\n");
2848 let has_text = !content.trim().is_empty();
2849 let has_tool_calls = !tool_calls.is_empty();
2850 // Reasoning replay must be a function of the stored message ONLY,
2851 // never of later history. DeepSeek's prefix cache hashes the raw
2852 // bytes of every message; flipping `reasoning_content` on/off
2853 // depending on whether a follow-up user turn exists rewrites a
2854 // historical message between turns and busts the cache from that
2855 // point onwards. Always emit `reasoning_content` when the model
2856 // requires replay AND the stored message carries thinking text.
2857 // Tool-call messages with empty thinking still need a placeholder
2858 // (DeepSeek 400s without it), but text-only assistant messages
2859 // simply omit the field when there's nothing to replay.
2860 let mut has_reasoning = include_reasoning && !reasoning_content.trim().is_empty();
2861 if include_reasoning && has_tool_calls && !has_reasoning {
2862 logging::warn(
2863 "Substituting placeholder reasoning_content for DeepSeek tool-call assistant message",
2864 );
2865 reasoning_content = String::from(REASONING_REPLAY_PLACEHOLDER);
2866 has_reasoning = true;
2867 }
2868
2869 // DeepSeek rejects assistant messages where both `content` and
2870 // `tool_calls` are missing/null. Skip such entries even if they
2871 // carry reasoning-only metadata unless we can send a non-null
2872 // placeholder content field.
2873 if !has_text && !has_tool_calls && !has_reasoning {
2874 pending_tool_calls.clear();
2875 deferred_tool_result_images.clear();
2876 continue;
2877 }
2878
2879 let mut msg = json!({
2880 "role": "assistant",
2881 "content": if has_text {
2882 json!(content)
2883 } else if has_reasoning {
2884 json!("")
2885 } else {
2886 Value::Null
2887 },
2888 });
2889 if has_reasoning {
2890 msg["reasoning_content"] = json!(reasoning_content);
2891 }
2892 if has_tool_calls {
2893 msg["tool_calls"] = json!(tool_calls);
2894 let expected_tool_result_count = tool_call_infos.len();
2895 pending_tool_calls = tool_call_infos.into_iter().collect();
2896 deferred_tool_result_images.clear();
2897 if pending_tool_calls.len() != expected_tool_result_count {
2898 logging::warn(
2899 "Rejecting assistant tool-call batch with duplicate tool_call IDs",
2900 );
2901 pending_tool_calls.clear();
2902 }
2903 } else {
2904 pending_tool_calls.clear();
2905 deferred_tool_result_images.clear();
2906 }
2907 out.push(msg);
2908 } else if matches!(placement, RolePlacement::System | RolePlacement::Developer) {
2909 let content = text_parts.join("\n");
2910 if !content.trim().is_empty() {
2911 let mut msg = json!({
2912 "role": if placement == RolePlacement::Developer {
2913 "developer"
2914 } else {
2915 "system"
2916 },
2917 "content": content,
2918 });
2919 if include_tool_budget_metadata && let Some(turn_meta) = &turn_meta_budget {
2920 msg["_turn_meta_budget"] = turn_meta_budget_json(turn_meta);
2921 }
2922 out.push(msg);
2923 }
2924 } else if placement == RolePlacement::User {
2925 let content = text_parts.join("\n");
2926 let has_text = !content.trim().is_empty();
2927 let has_images = !image_parts.is_empty();
2928 if has_text || has_images {
2929 let wire_content = if has_images {
2930 let mut parts = Vec::new();
2931 if has_text {
2932 parts.push(json!({
2933 "type": "text",
2934 "text": content,
2935 }));
2936 }
2937 parts.extend(image_parts);
2938 json!(parts)
2939 } else {
2940 json!(content)
2941 };
2942 let mut msg = json!({
2943 "role": "user",
2944 "content": wire_content,
2945 });
2946 if include_tool_budget_metadata && let Some(turn_meta) = &turn_meta_budget {
2947 msg["_turn_meta_budget"] = turn_meta_budget_json(turn_meta);
2948 }
2949 if (Some(message_index) == summary_index
2950 || Some(message_index) == summary_target
2951 || crate::compaction::is_wire_compaction_checkpoint_message(message)
2952 || crate::runtime_handoff::is_agent_topology_checkpoint(message)
2953 || crate::runtime_handoff::is_restored_agent_topology_checkpoint(message))
2954 && let Some(previous) = out.last_mut()
2955 && previous.get("role").and_then(Value::as_str) == Some("user")
2956 {
2957 let previous_content = previous["content"].take();
2958 let current_content = msg["content"].take();
2959 previous["content"] =
2960 merge_adjacent_user_content(previous_content, current_content);
2961 if previous.get("_turn_meta_budget").is_none()
2962 && let Some(meta) = msg.get("_turn_meta_budget")
2963 {
2964 previous["_turn_meta_budget"] = meta.clone();
2965 }
2966 } else {
2967 out.push(msg);
2968 }
2969 }
2970 }
2971
2972 // A user/system/developer wire message closes the contiguous run that
2973 // must follow an assistant tool-call message. If the same stored
2974 // message also carries a later tool result, reject it here instead of
2975 // briefly accepting the result and letting any synthesized media
2976 // escape after the safety pass strips the malformed batch.
2977 if out.len() > out_len_before_role_projection
2978 && !placement.is_assistant_channel()
2979 && !pending_tool_calls.is_empty()
2980 {
2981 logging::warn("Dropping tool-call batch interrupted by non-tool content");
2982 pending_tool_calls.clear();
2983 deferred_tool_result_images.clear();
2984 }
2985
2986 if !tool_results.is_empty() {
2987 if pending_tool_calls.is_empty() {
2988 logging::warn("Dropping tool results without matching tool_calls");
2989 } else {
2990 for (tool_id, content, message_label, content_blocks) in tool_results {
2991 if let Some(tool_info) = pending_tool_calls.remove(&tool_id) {
2992 let (image, omitted) = crate::image_attach::provider_tool_result_image_refs(
2993 Some(&content_blocks),
2994 );
2995 let content =
2996 crate::image_attach::tool_result_text_with_omission(&content, omitted);
2997 let wire_result = compact_tool_result_for_wire(
2998 &tool_info.tool_name,
2999 &tool_info.input,
3000 &content,
3001 &message_label,
3002 tool_result_budget,
3003 &mut seen_tool_results,
3004 );
3005 let mut tool_msg = json!({
3006 "role": "tool",
3007 "tool_call_id": tool_id,
3008 "content": wire_result.content,
3009 });
3010 if include_tool_budget_metadata {
3011 tool_msg["_tool_result_budget"] = json!({
3012 "original_chars": wire_result.original_chars,
3013 "sent_chars": wire_result.sent_chars,
3014 "truncated": wire_result.truncated,
3015 "deduplicated": wire_result.deduplicated,
3016 });
3017 }
3018 out.push(tool_msg);
3019 if let Some((mime_type, data)) = image {
3020 deferred_tool_result_images.push(json!({
3021 "type": "text",
3022 "text": format!(
3023 "Image returned by tool `{}` (call `{tool_id}`):",
3024 tool_info.tool_name,
3025 ),
3026 }));
3027 deferred_tool_result_images.push(json!({
3028 "type": "image_url",
3029 "image_url": {
3030 "url": format!("data:{mime_type};base64,{data}")
3031 },
3032 }));
3033 }
3034 } else {
3035 logging::warn(format!(
3036 "Dropping tool result for unknown tool_call_id: {tool_id}"
3037 ));
3038 }
3039 }
3040 if pending_tool_calls.is_empty() && !deferred_tool_result_images.is_empty() {
3041 out.push(json!({
3042 "role": "user",
3043 "content": std::mem::take(&mut deferred_tool_result_images),
3044 }));
3045 }
3046 }
3047 } else if !placement.is_assistant_channel() {
3048 pending_tool_calls.clear();
3049 deferred_tool_result_images.clear();
3050 }
3051 }
3052
3053 // Safety net: after compaction, an assistant message may have tool_calls
3054 // whose results were summarized away. The API rejects these, so strip
3055 // the tool_calls (downgrading to a plain assistant message) and remove
3056 // the now-orphaned tool result messages.
3057 let mut i = 0;
3058 while i < out.len() {
3059 let is_assistant_with_tools = out[i].get("role").and_then(Value::as_str)
3060 == Some("assistant")
3061 && out[i].get("tool_calls").is_some();
3062
3063 if is_assistant_with_tools {
3064 let expected_ids: Vec<String> = out[i]
3065 .get("tool_calls")
3066 .and_then(Value::as_array)
3067 .map(|calls| {
3068 calls
3069 .iter()
3070 .filter_map(|c| c.get("id").and_then(Value::as_str).map(String::from))
3071 .collect()
3072 })
3073 .unwrap_or_default();
3074
3075 // Collect tool result IDs immediately following this assistant message.
3076 let mut found_ids = Vec::new();
3077 let mut tool_result_end = i + 1;
3078 while tool_result_end < out.len() {
3079 if out[tool_result_end].get("role").and_then(Value::as_str) == Some("tool") {
3080 if let Some(id) = out[tool_result_end]
3081 .get("tool_call_id")
3082 .and_then(Value::as_str)
3083 {
3084 found_ids.push(id.to_string());
3085 }
3086 tool_result_end += 1;
3087 } else {
3088 break;
3089 }
3090 }
3091
3092 // Chat Completions accepts only the immediately contiguous tool
3093 // run after its assistant tool-call message. Do not accept a
3094 // later tool result after user/system content has intervened.
3095 let results_match = expected_ids.len() == found_ids.len()
3096 && expected_ids.iter().all(|id| found_ids.contains(id));
3097 if !results_match {
3098 let missing: Vec<_> = expected_ids
3099 .iter()
3100 .filter(|id| !found_ids.contains(*id))
3101 .collect();
3102 logging::warn(format!(
3103 "Stripping orphaned tool_calls from assistant message \
3104 (expected {} tool results, found {}, missing: {:?})",
3105 expected_ids.len(),
3106 found_ids.len(),
3107 missing
3108 ));
3109 if let Some(obj) = out[i].as_object_mut() {
3110 obj.remove("tool_calls");
3111 }
3112 // If tool_calls were the only assistant content, remove the now-invalid
3113 // assistant message entirely (DeepSeek requires content or tool_calls).
3114 let assistant_content_empty = out[i]
3115 .get("content")
3116 .is_none_or(|v| v.is_null() || v.as_str().is_some_and(str::is_empty));
3117 if assistant_content_empty {
3118 // Remove orphaned tool results tied to this stripped assistant call set.
3119 let mut j = out.len();
3120 while j > i + 1 {
3121 j -= 1;
3122 if out[j].get("role").and_then(Value::as_str) == Some("tool")
3123 && let Some(id) = out[j].get("tool_call_id").and_then(Value::as_str)
3124 && expected_ids.iter().any(|expected| expected == id)
3125 {
3126 out.remove(j);
3127 }
3128 }
3129 out.remove(i);
3130 i = i.saturating_sub(1);
3131 continue;
3132 }
3133 // Remove contiguous tool results first
3134 if tool_result_end > i + 1 {
3135 out.drain((i + 1)..tool_result_end);
3136 }
3137 // Remove any remaining non-contiguous tool results referencing expected_ids
3138 // (scan backward to avoid index shifting issues)
3139 let mut j = out.len();
3140 while j > i + 1 {
3141 j -= 1;
3142 if out[j].get("role").and_then(Value::as_str) == Some("tool")
3143 && let Some(id) = out[j].get("tool_call_id").and_then(Value::as_str)
3144 && expected_ids.iter().any(|expected| expected == id)
3145 {
3146 out.remove(j);
3147 }
3148 }
3149 }
3150 }
3151 i += 1;
3152 }
3153
3154 out
3155 }
3156
3157 pub(super) fn tool_to_chat(tool: &Tool) -> Value {
3158 let mut value = json!({
3159 "type": "function",
3160 "function": {
3161 "name": to_api_tool_name(&tool.name),
3162 "description": tool.description,
3163 "parameters": tool.input_schema,
3164 }
3165 });
3166 if let Some(strict) = tool.strict
3167 && let Some(function) = value.get_mut("function")
3168 {
3169 function["strict"] = json!(strict);
3170 }
3171 value
3172 }
3173
3174 pub(super) fn tool_to_chat_for_base_url(tool: &Tool, base_url: &str) -> Value {
3175 let mut value = tool_to_chat(tool);
3176 if !deepseek_base_url_supports_strict_tools(base_url)
3177 && let Some(function) = value.get_mut("function")
3178 && let Some(obj) = function.as_object_mut()
3179 {
3180 obj.remove("strict");
3181 }
3182 value
3183 }
3184
3185 fn deepseek_base_url_supports_strict_tools(base_url: &str) -> bool {
3186 let trimmed = base_url.trim_end_matches('/').to_ascii_lowercase();
3187 let is_deepseek = trimmed == "https://api.deepseek.com"
3188 || trimmed == "https://api.deepseek.com/v1"
3189 || trimmed == "https://api.deepseek.com/beta"
3190 || trimmed == "https://api.deepseeki.com"
3191 || trimmed == "https://api.deepseeki.com/v1"
3192 || trimmed == "https://api.deepseeki.com/beta";
3193 !is_deepseek || trimmed.ends_with("/beta")
3194 }
3195
3196 fn map_tool_choice_for_chat(choice: &Value) -> Option<Value> {
3197 if let Some(choice_str) = choice.as_str() {
3198 return Some(json!(choice_str));
3199 }
3200 let Some(choice_type) = choice.get("type").and_then(Value::as_str) else {
3201 return Some(choice.clone());
3202 };
3203
3204 match choice_type {
3205 "auto" | "none" => Some(json!(choice_type)),
3206 "any" => Some(json!("auto")),
3207 "tool" => choice.get("name").and_then(Value::as_str).map(|name| {
3208 json!({
3209 "type": "function",
3210 "function": { "name": to_api_tool_name(name) }
3211 })
3212 }),
3213 _ => Some(choice.clone()),
3214 }
3215 }
3216
3217 fn should_send_tool_choice_for_chat(provider: ProviderKind, effort: Option<&str>) -> bool {
3218 if !matches!(provider, ProviderKind::Deepseek) {
3219 return true;
3220 }
3221 !reasoning_effort_enables_thinking(effort)
3222 }
3223
3224 fn reasoning_effort_enables_thinking(effort: Option<&str>) -> bool {
3225 let Some(effort) = effort else {
3226 return false;
3227 };
3228 !matches!(
3229 effort.trim().to_ascii_lowercase().as_str(),
3230 "off" | "disabled" | "none" | "false"
3231 )
3232 }
3233
3234 /// Final-pass sanitizer over the outgoing chat-completions JSON payload.
3235 /// Forces a non-empty `reasoning_content` onto assistant messages that carry
3236 /// `tool_calls`, when the model + effort combination requires it. DeepSeek's
3237 /// thinking-mode API rejects such messages with a 400 error; substituting a
3238 /// placeholder keeps the conversation chain intact. Non-tool assistant
3239 /// reasoning can stay omitted once a later user text turn begins.
3240 ///
3241 /// Also tallies the size of all replayed `reasoning_content` and logs it, so
3242 /// users on `RUST_LOG=codewhale_tui=debug` can see how much of their input
3243 /// budget is being spent re-sending prior thinking traces.
3244 #[cfg(test)]
3245 pub(super) fn sanitize_thinking_mode_messages(
3246 body: &mut Value,
3247 model: &str,
3248 effort: Option<&str>,
3249 provider: ProviderKind,
3250 ) -> Option<u32> {
3251 sanitize_thinking_mode_messages_for_route(body, model, effort, provider, "")
3252 }
3253
3254 /// Route-aware variant of `sanitize_thinking_mode_messages`.
3255 ///
3256 /// The wrapper above remains intentionally route-agnostic for existing test
3257 /// helpers and generic callers. Production chat requests call this version so
3258 /// exact Kimi Code K3 assistant tool turns retain the reasoning trace that
3259 /// K3 expects on the next request.
3260 pub(super) fn sanitize_thinking_mode_messages_for_route(
3261 body: &mut Value,
3262 model: &str,
3263 effort: Option<&str>,
3264 provider: ProviderKind,
3265 base_url: &str,
3266 ) -> Option<u32> {
3267 // Mistral replay is encoded inside polymorphic `content` blocks, not the
3268 // DeepSeek `reasoning_content` field. Running the DeepSeek placeholder
3269 // sanitizer after reshaping would add a second, invalid reasoning dialect
3270 // to assistant tool-call turns.
3271 if is_exact_mistral_chat_route(provider, base_url) {
3272 return None;
3273 }
3274 if !should_replay_reasoning_content_for_provider_on_route(provider, base_url, model, effort) {
3275 return None;
3276 }
3277 let messages = body.get_mut("messages").and_then(Value::as_array_mut)?;
3278 let mut substitutions: u32 = 0;
3279 let mut replay_chars: u64 = 0;
3280 let mut replay_messages: u32 = 0;
3281 for (idx, msg) in messages.iter_mut().enumerate() {
3282 if msg.get("role").and_then(Value::as_str) != Some("assistant") {
3283 continue;
3284 }
3285 let has_tool_calls = msg.get("tool_calls").is_some();
3286 let needs_placeholder = msg
3287 .get("reasoning_content")
3288 .and_then(Value::as_str)
3289 .is_none_or(|s| s.trim().is_empty());
3290 if has_tool_calls && needs_placeholder {
3291 msg["reasoning_content"] = json!(REASONING_REPLAY_PLACEHOLDER);
3292 substitutions = substitutions.saturating_add(1);
3293 logging::warn(format!(
3294 "Final sanitizer: forced reasoning_content placeholder on assistant[{idx}]",
3295 ));
3296 }
3297 if let Some(reasoning) = msg.get("reasoning_content").and_then(Value::as_str) {
3298 let len = reasoning.len() as u64;
3299 if len > 0 {
3300 replay_chars = replay_chars.saturating_add(len);
3301 replay_messages = replay_messages.saturating_add(1);
3302 }
3303 }
3304 }
3305 if substitutions > 0 {
3306 logging::warn(format!(
3307 "Final sanitizer: {substitutions} assistant message(s) needed reasoning_content placeholder",
3308 ));
3309 }
3310 if replay_messages == 0 {
3311 return None;
3312 }
3313 // ~4 chars/token is the standard rough estimate; DeepSeek tokens skew
3314 // a touch shorter on Chinese/code but this is order-of-magnitude info.
3315 let approx_tokens = (replay_chars / 4).min(u64::from(u32::MAX)) as u32;
3316 logging::info(format!(
3317 "Reasoning-content replay: {replay_messages} assistant message(s), ~{approx_tokens} input tokens ({replay_chars} chars) being re-sent in this request",
3318 ));
3319 Some(approx_tokens)
3320 }
3321
3322 /// Sums the byte length of `reasoning_content` across all assistant messages in
3323 /// an outgoing chat-completions body. Used by tests; the production sanitizer
3324 /// computes the same number inline and logs it.
3325 #[cfg(test)]
3326 pub(super) fn count_reasoning_replay_chars(body: &Value) -> u64 {
3327 let Some(messages) = body.get("messages").and_then(Value::as_array) else {
3328 return 0;
3329 };
3330 messages
3331 .iter()
3332 .filter(|m| m.get("role").and_then(Value::as_str) == Some("assistant"))
3333 .filter_map(|m| m.get("reasoning_content").and_then(Value::as_str))
3334 .map(|s| s.len() as u64)
3335 .sum()
3336 }
3337
3338 /// Render the transport-shape headers we care about for #103 diagnostics.
3339 /// Always returns SOMETHING printable so the decode-error log line is parseable
3340 /// even when the server stripped a header we expected.
3341 fn format_stream_headers(headers: &reqwest::header::HeaderMap) -> String {
3342 const FIELDS: &[&str] = &[
3343 "content-encoding",
3344 "transfer-encoding",
3345 "connection",
3346 "server",
3347 ];
3348 let mut parts: Vec<String> = Vec::with_capacity(FIELDS.len());
3349 for field in FIELDS {
3350 let rendered = headers
3351 .get(*field)
3352 .and_then(|v| v.to_str().ok())
3353 .unwrap_or("(absent)");
3354 parts.push(format!("{field}={rendered}"));
3355 }
3356 parts.join(", ")
3357 }
3358
3359 /// Diagnostic logger fired when DeepSeek rejects the request despite the
3360 /// sanitizer. Walks the body and logs which assistant messages have tool_calls
3361 /// but no `reasoning_content` — useful to track down a code path that bypasses
3362 /// the sanitizer entirely.
3363 fn log_thinking_mode_violations(body: &Value) {
3364 let Some(messages) = body.get("messages").and_then(Value::as_array) else {
3365 logging::warn("400-after-sanitizer: body has no `messages` array");
3366 return;
3367 };
3368 let mut violations: Vec<String> = Vec::new();
3369 for (idx, msg) in messages.iter().enumerate() {
3370 if msg.get("role").and_then(Value::as_str) != Some("assistant") {
3371 continue;
3372 }
3373 let reasoning = msg
3374 .get("reasoning_content")
3375 .and_then(Value::as_str)
3376 .unwrap_or("");
3377 let has_tc = msg.get("tool_calls").is_some();
3378 if reasoning.trim().is_empty() {
3379 violations.push(format!(
3380 "assistant[{idx}] (reasoning_content missing, tool_calls={has_tc})"
3381 ));
3382 }
3383 }
3384 if violations.is_empty() {
3385 logging::warn(
3386 "400-after-sanitizer: all assistant messages have reasoning_content — DeepSeek rejected for a different reason",
3387 );
3388 } else {
3389 logging::warn(format!(
3390 "400-after-sanitizer: {} assistant message(s) lack reasoning_content despite sanitizer: {}",
3391 violations.len(),
3392 violations.join(", ")
3393 ));
3394 }
3395 }
3396
3397 fn requires_reasoning_content(model: &str) -> bool {
3398 let lower = model.to_lowercase();
3399 // V4-family direct model IDs.
3400 lower.contains("deepseek-v4")
3401 // Public DeepSeek API aliases routed server-side to the V4 family.
3402 // `deepseek-chat` resolves to `deepseek-v4-flash` and `deepseek-reasoner`
3403 // resolves to `deepseek-v4-pro`; both have thinking mode enabled by
3404 // default, so any assistant message carrying tool_calls must replay
3405 // `reasoning_content` on subsequent turns or the API returns 400.
3406 || lower.starts_with("deepseek-chat")
3407 || lower.starts_with("deepseek-reasoner")
3408 || has_deepseek_r_series_marker(&lower)
3409 // #6044: the V4.1 official id dropped the version number entirely
3410 // (`deepseek-flash`), so the literal arms above cannot see it and
3411 // the decode/replay classifiers depended on a later catalog fallback
3412 // to catch it. The catalog owns the capability — consult it here so
3413 // every caller (stream style, wire replay, prompt inspection) agrees
3414 // without another hardcoded id.
3415 || (lower.starts_with("deepseek-") && model_supports_reasoning(model))
3416 }
3417
3418 fn should_replay_reasoning_content(model: &str, effort: Option<&str>) -> bool {
3419 if effort
3420 .map(|value| {
3421 matches!(
3422 value.trim().to_ascii_lowercase().as_str(),
3423 "off" | "disabled" | "none" | "false"
3424 )
3425 })
3426 .unwrap_or(false)
3427 {
3428 return false;
3429 }
3430
3431 requires_reasoning_content(model)
3432 }
3433
3434 #[cfg(test)]
3435 fn should_replay_reasoning_content_for_provider(
3436 provider: ProviderKind,
3437 model: &str,
3438 effort: Option<&str>,
3439 ) -> bool {
3440 should_replay_reasoning_content_for_provider_on_route(provider, "", model, effort)
3441 }
3442
3443 /// Route-aware reasoning replay policy.
3444 ///
3445 /// Keep the bare K3 identifier out of the global model catalog: direct
3446 /// Moonshot and arbitrary OpenAI-compatible routes can also expose a `k3`
3447 /// model name, but only Kimi Code's exact membership-plan endpoint has this
3448 /// replay contract.
3449 fn should_replay_reasoning_content_for_provider_on_route(
3450 provider: ProviderKind,
3451 base_url: &str,
3452 model: &str,
3453 effort: Option<&str>,
3454 ) -> bool {
3455 // Exact always-thinking routes replay their reasoning trace regardless of
3456 // a stale caller effort: the API contract requires the assistant
3457 // reasoning field on later tool turns for multi-turn continuity.
3458 if is_exact_direct_moonshot_k3_route(provider, base_url, model)
3459 || is_exact_kimi_code_k3_route(provider, base_url, model)
3460 || (is_exact_mistral_chat_route(provider, base_url)
3461 && mistral_model_has_native_reasoning(model))
3462 {
3463 return true;
3464 }
3465
3466 // The exact Model Studio route policy evaluates BEFORE any generic
3467 // model-name heuristic. Only models Alibaba documents as accepting
3468 // `preserve_thinking` replay historical `reasoning_content`, plus the
3469 // concrete DeepSeek V4 family ids whose own API contract requires the
3470 // reasoning field on tool turns. A model named `foo-thinking` or
3471 // `foo-reasoner` proves nothing about DashScope's request dialect and
3472 // must not gain replay here — replaying stale Thinking blocks feeds the
3473 // model its own past reasoning and re-triggers it every turn (observed
3474 // as a repeated handoff loop with the always-thinking qwen3.8 family).
3475 // Pi does not replay those either.
3476 if is_exact_modelstudio_chat_route(provider, base_url) {
3477 if modelstudio_model_supports_preserve_thinking(model) {
3478 // Thinking-only preserve models (kimi-k2.7-code) are
3479 // always-thinking and replay even with a stale `off`; hybrid
3480 // preserve models defer to the effort gate like every other
3481 // hybrid.
3482 if is_exact_modelstudio_thinking_only_route(provider, base_url, model)
3483 || !modelstudio_effort_disables_thinking(effort)
3484 {
3485 return true;
3486 }
3487 return false;
3488 }
3489 let lower = model.trim().to_ascii_lowercase();
3490 return lower.contains("deepseek-v4")
3491 || lower.starts_with("deepseek-chat")
3492 || lower.starts_with("deepseek-reasoner");
3493 }
3494
3495 if effort
3496 .map(|value| {
3497 matches!(
3498 value.trim().to_ascii_lowercase().as_str(),
3499 "off" | "disabled" | "none" | "false"
3500 )
3501 })
3502 .unwrap_or(false)
3503 {
3504 return false;
3505 }
3506
3507 if requires_reasoning_content(model) {
3508 return true;
3509 }
3510
3511 // Xiaomi MiMo's API requires the assistant `reasoning_content` back on
3512 // tool-call turns (omitting it is a 400), for every MiMo chat model, not
3513 // only the ids the offline catalog marks as reasoning (#6501). A turn
3514 // that produced no reasoning replays nothing.
3515 if provider == ProviderKind::XiaomiMimo {
3516 return true;
3517 }
3518
3519 if is_exact_mistral_chat_route(provider, base_url)
3520 && mistral_model_has_adjustable_reasoning(model)
3521 {
3522 return true;
3523 }
3524
3525 if !provider_accepts_reasoning_content(provider) {
3526 // Generic non-DeepSeek model on a provider that rejects the field:
3527 // keep stripping it (preserves the #1542 fix). But a known DeepSeek
3528 // reasoning model pointed at a DeepSeek-compatible endpoint via the
3529 // generic `openai` provider still requires reasoning_content replay,
3530 // or the thinking-mode API returns 400 (#1739 / #1694).
3531 return false;
3532 }
3533
3534 model_supports_reasoning(model)
3535 }
3536
3537 /// Test shorthand: does this route surface `reasoning_content` /
3538 /// `reasoning` / `reasoning_details` deltas as Thinking by default?
3539 #[cfg(test)]
3540 fn is_reasoning_model_for_stream(provider: ProviderKind, model: &str) -> bool {
3541 reasoning_stream_style_for_stream(provider, model, None) == ReasoningStreamStyle::SeparateField
3542 }
3543
3544 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
3545 pub(super) enum ReasoningStreamStyle {
3546 SeparateField,
3547 InlineTags,
3548 MistralBlocks,
3549 None,
3550 }
3551
3552 #[cfg(test)]
3553 fn reasoning_stream_style_for_stream(
3554 provider: ProviderKind,
3555 model: &str,
3556 configured: Option<&str>,
3557 ) -> ReasoningStreamStyle {
3558 reasoning_stream_style_for_route(provider, "", model, configured)
3559 }
3560
3561 /// Choose stream decoding semantics for a fully resolved provider route.
3562 fn reasoning_stream_style_for_route(
3563 provider: ProviderKind,
3564 base_url: &str,
3565 model: &str,
3566 configured: Option<&str>,
3567 ) -> ReasoningStreamStyle {
3568 if is_exact_mistral_chat_route(provider, base_url) && mistral_model_supports_reasoning(model) {
3569 return ReasoningStreamStyle::MistralBlocks;
3570 }
3571 if let Some(configured) = configured {
3572 if let Some(style) = parse_reasoning_stream_style(configured) {
3573 return style;
3574 }
3575 logging::warn(format!(
3576 "Ignoring unrecognized reasoning_stream_style `{configured}`; expected separate_field, inline_tags, or none"
3577 ));
3578 }
3579 // #6501: the dedicated reasoning fields are reasoning on every
3580 // Chat Completions route that sends them (DeepSeek, xAI, Xiaomi MiMo,
3581 // Kimi, GLM, MiniMax, DashScope, vLLM/SGLang reasoning parsers, OpenRouter,
3582 // Ollama, ...), and the non-streaming parser already treats them that way
3583 // unconditionally. Gating the stream on a provider allowlist AND a
3584 // per-model catalog row meant every unlisted provider (xAI, StepFun,
3585 // Together, Ollama, the Codewhale gateway, ...) and every model id newer
3586 // than the offline catalog (grok-4.7, a fresh MiMo id) rendered its
3587 // reasoning as answer prose. Surfacing a field that never arrives costs
3588 // nothing; a gateway that really puts its answer in `reasoning_content`
3589 // opts out with `reasoning_stream_style = "none"`. Replay of reasoning in
3590 // request history stays separately gated by
3591 // `should_replay_reasoning_content_for_provider_on_route`.
3592 ReasoningStreamStyle::SeparateField
3593 }
3594
3595 fn parse_reasoning_stream_style(value: &str) -> Option<ReasoningStreamStyle> {
3596 match value.trim().to_ascii_lowercase().replace('-', "_").as_str() {
3597 "separate_field" | "separate" | "field" => Some(ReasoningStreamStyle::SeparateField),
3598 "inline_tags" | "inline" | "think_tags" | "thinking_tags" => {
3599 Some(ReasoningStreamStyle::InlineTags)
3600 }
3601 "none" | "text" | "disabled" | "off" => Some(ReasoningStreamStyle::None),
3602 _ => None,
3603 }
3604 }
3605
3606 /// Providers whose chat-completions API both returns and accepts a dedicated
3607 /// `reasoning_content` field on assistant messages.
3608 ///
3609 /// Arcee is intentionally included. Trinity-Large-Thinking natively emits
3610 /// `<think>...</think>` traces, but Arcee's hosted API serves it through vLLM
3611 /// with `--reasoning-parser deepseek_r1`, which parses those blocks into a
3612 /// `reasoning_content` field (verified live against `api.arcee.ai`: thinking
3613 /// streams as `delta.reasoning_content`, the answer as `delta.content`, with no
3614 /// `<think>` tags on the wire). Arcee's docs require replaying `reasoning_content`
3615 /// on assistant tool-call turns; dropping it makes the model emit tool calls as
3616 /// raw XML inside its thinking ("xml_in_reasoning" pitfall). Do not remove Arcee
3617 /// here without new live evidence — see docs.arcee.ai/capabilities/reasoning-traces.
3618 fn provider_accepts_reasoning_content(provider: ProviderKind) -> bool {
3619 matches!(
3620 provider,
3621 ProviderKind::Deepseek
3622 | ProviderKind::NvidiaNim
3623 | ProviderKind::Openrouter
3624 | ProviderKind::XiaomiMimo
3625 | ProviderKind::Novita
3626 | ProviderKind::Fireworks
3627 | ProviderKind::Siliconflow
3628 | ProviderKind::SiliconflowCN
3629 | ProviderKind::Volcengine
3630 | ProviderKind::Arcee
3631 | ProviderKind::Minimax
3632 | ProviderKind::Sglang
3633 | ProviderKind::Zai
3634 | ProviderKind::Moonshot // #3016: Kimi thinking traces use reasoning_content
3635 )
3636 }
3637
3638 fn has_deepseek_r_series_marker(model_lower: &str) -> bool {
3639 const PREFIX: &str = "deepseek-r";
3640 model_lower.match_indices(PREFIX).any(|(idx, _)| {
3641 model_lower[idx + PREFIX.len()..]
3642 .chars()
3643 .next()
3644 .is_some_and(|ch| ch.is_ascii_digit())
3645 })
3646 }
3647
3648 /// Transport-only reasoning replay placeholder. DeepSeek-family chat wires
3649 /// reject assistant tool-call messages with an empty `reasoning_content`, so
3650 /// the request serializer substitutes this string for replay only. Providers
3651 /// that mirror assistant history (GLM-5.x) stream the substituted field back
3652 /// as a live reasoning delta; ingest must drop that exact echo so a wire-only
3653 /// placeholder never becomes a persisted or displayed thinking block.
3654 pub(crate) const REASONING_REPLAY_PLACEHOLDER: &str = "(reasoning omitted)";
3655
3656 #[must_use]
3657 pub(crate) fn is_reasoning_replay_placeholder(text: &str) -> bool {
3658 text.trim() == REASONING_REPLAY_PLACEHOLDER
3659 }
3660
3661 fn reasoning_delta(
3662 value: &Value,
3663 choice_index: u32,
3664 reasoning_detail_buffers: &mut std::collections::HashMap<u32, String>,
3665 ) -> Option<String> {
3666 if let Some(reasoning) = value
3667 .get("reasoning_content")
3668 .or_else(|| value.get("reasoning"))
3669 .and_then(Value::as_str)
3670 {
3671 if is_reasoning_replay_placeholder(reasoning) {
3672 return None;
3673 }
3674 return Some(reasoning.to_string());
3675 }
3676
3677 let details = value.get("reasoning_details").and_then(Value::as_array)?;
3678 let full_text = details
3679 .iter()
3680 .filter_map(|detail| detail.get("text").and_then(Value::as_str))
3681 .collect::<String>();
3682 if full_text.is_empty() {
3683 return None;
3684 }
3685
3686 let previous = reasoning_detail_buffers.entry(choice_index).or_default();
3687 let delta = full_text
3688 .strip_prefix(previous.as_str())
3689 .unwrap_or(&full_text)
3690 .to_string();
3691 *previous = full_text;
3692 Some(delta)
3693 }
3694
3695 fn reasoning_message_text(value: &Value) -> Option<String> {
3696 if let Some(reasoning) = value
3697 .get("reasoning_content")
3698 .or_else(|| value.get("reasoning"))
3699 .and_then(Value::as_str)
3700 {
3701 if is_reasoning_replay_placeholder(reasoning) {
3702 return None;
3703 }
3704 return Some(reasoning.to_string());
3705 }
3706 value
3707 .get("reasoning_details")
3708 .and_then(Value::as_array)
3709 .map(|details| {
3710 details
3711 .iter()
3712 .filter_map(|detail| detail.get("text").and_then(Value::as_str))
3713 .collect::<String>()
3714 })
3715 }
3716
3717 #[cfg(test)]
3718 pub(super) fn parse_chat_message(payload: &Value) -> Result<MessageResponse> {
3719 parse_chat_message_for_route(payload, ProviderKind::Openai, "")
3720 }
3721
3722 fn parse_chat_message_for_route(
3723 payload: &Value,
3724 provider: ProviderKind,
3725 base_url: &str,
3726 ) -> Result<MessageResponse> {
3727 let id = payload
3728 .get("id")
3729 .and_then(Value::as_str)
3730 .unwrap_or("chatcmpl")
3731 .to_string();
3732 let model = payload
3733 .get("model")
3734 .and_then(Value::as_str)
3735 .unwrap_or("unknown")
3736 .to_string();
3737
3738 let choices = payload
3739 .get("choices")
3740 .and_then(Value::as_array)
3741 .context("Chat API response missing choices")?;
3742 let choice = choices
3743 .first()
3744 .context("Chat API response missing first choice")?;
3745 let message = choice
3746 .get("message")
3747 .context("Chat API response missing message")?;
3748
3749 let mut content_blocks = Vec::new();
3750 if let Some(reasoning) =
3751 reasoning_message_text(message).filter(|reasoning| !reasoning.trim().is_empty())
3752 {
3753 content_blocks.push(ContentBlock::Thinking {
3754 signature: None,
3755 state: None,
3756 thinking: reasoning.to_string(),
3757 });
3758 }
3759 let (mistral_thinking, mistral_text) = if is_exact_mistral_chat_route(provider, base_url) {
3760 extract_mistral_polymorphic_content(message)
3761 } else {
3762 (None, None)
3763 };
3764 if let Some(thinking) = mistral_thinking.filter(|s| !s.trim().is_empty()) {
3765 content_blocks.push(ContentBlock::Thinking {
3766 signature: None,
3767 state: None,
3768 thinking,
3769 });
3770 }
3771 if let Some(text) = mistral_text.filter(|s| !s.trim().is_empty()) {
3772 content_blocks.push(ContentBlock::Text {
3773 text,
3774 cache_control: None,
3775 });
3776 } else if let Some(text) = message.get("content").and_then(Value::as_str)
3777 && !text.trim().is_empty()
3778 {
3779 content_blocks.push(ContentBlock::Text {
3780 text: text.to_string(),
3781 cache_control: None,
3782 });
3783 }
3784
3785 if let Some(tool_calls) = message.get("tool_calls").and_then(Value::as_array) {
3786 for call in tool_calls {
3787 let id = call
3788 .get("id")
3789 .and_then(Value::as_str)
3790 .unwrap_or("tool_call")
3791 .to_string();
3792 let function = call.get("function");
3793 let name = tool_name_or_fallback(
3794 function.and_then(|f| f.get("name")).and_then(Value::as_str),
3795 &id,
3796 "Non-streaming response",
3797 );
3798 let arguments = function
3799 .and_then(|f| f.get("arguments"))
3800 .and_then(Value::as_str)
3801 .map(|raw| serde_json::from_str(raw).unwrap_or(Value::String(raw.to_string())))
3802 .unwrap_or(Value::Null);
3803 let caller = call.get("caller").and_then(|v| {
3804 v.get("type")
3805 .and_then(Value::as_str)
3806 .map(|caller_type| ToolCaller {
3807 caller_type: caller_type.to_string(),
3808 tool_id: v
3809 .get("tool_id")
3810 .and_then(Value::as_str)
3811 .map(std::string::ToString::to_string),
3812 })
3813 });
3814
3815 let thought_signature = call
3816 .pointer("/extra_content/google/thought_signature")
3817 .and_then(Value::as_str)
3818 .map(str::to_string);
3819 content_blocks.push(ContentBlock::ToolUse {
3820 execution_id: None,
3821 id,
3822 name: from_api_tool_name(&name),
3823 input: arguments,
3824 caller,
3825 thought_signature,
3826 });
3827 }
3828 }
3829
3830 let usage = parse_usage(payload.get("usage"));
3831
3832 Ok(MessageResponse {
3833 id,
3834 r#type: "message".to_string(),
3835 role: "assistant".to_string(),
3836 content: content_blocks,
3837 model,
3838 stop_reason: choice
3839 .get("finish_reason")
3840 .and_then(Value::as_str)
3841 .map(str::to_string),
3842 stop_sequence: None,
3843 container: None,
3844 usage,
3845 })
3846 }
3847
3848 #[derive(Debug, Default)]
3849 struct InlineReasoningTagState {
3850 inside_think: bool,
3851 pending: String,
3852 }
3853
3854 #[derive(Debug, PartialEq, Eq)]
3855 enum ReasoningSegment {
3856 Text(String),
3857 Thinking(String),
3858 }
3859
3860 fn inline_reasoning_segments(
3861 content: &str,
3862 state: &mut InlineReasoningTagState,
3863 flush: bool,
3864 ) -> Vec<ReasoningSegment> {
3865 state.pending.push_str(content);
3866 let mut segments = Vec::new();
3867
3868 loop {
3869 if state.pending.is_empty() {
3870 break;
3871 }
3872
3873 if state.inside_think {
3874 if let Some(close_at) = state.pending.find("</think>") {
3875 push_reasoning_segment(
3876 &mut segments,
3877 ReasoningSegment::Thinking(state.pending[..close_at].to_string()),
3878 );
3879 state.pending.drain(..close_at + "</think>".len());
3880 state.inside_think = false;
3881 continue;
3882 }
3883
3884 let hold_len = if flush {
3885 0
3886 } else {
3887 trailing_tag_prefix_len(&state.pending, "</think>")
3888 };
3889 let emit_len = state.pending.len().saturating_sub(hold_len);
3890 if emit_len > 0 {
3891 push_reasoning_segment(
3892 &mut segments,
3893 ReasoningSegment::Thinking(state.pending[..emit_len].to_string()),
3894 );
3895 state.pending.drain(..emit_len);
3896 }
3897 break;
3898 }
3899
3900 if let Some(open_at) = state.pending.find("<think>") {
3901 push_reasoning_segment(
3902 &mut segments,
3903 ReasoningSegment::Text(state.pending[..open_at].to_string()),
3904 );
3905 state.pending.drain(..open_at + "<think>".len());
3906 state.inside_think = true;
3907 continue;
3908 }
3909
3910 let hold_len = if flush {
3911 0
3912 } else {
3913 trailing_tag_prefix_len(&state.pending, "<think>")
3914 };
3915 let emit_len = state.pending.len().saturating_sub(hold_len);
3916 if emit_len > 0 {
3917 push_reasoning_segment(
3918 &mut segments,
3919 ReasoningSegment::Text(state.pending[..emit_len].to_string()),
3920 );
3921 state.pending.drain(..emit_len);
3922 }
3923 break;
3924 }
3925
3926 segments
3927 }
3928
3929 fn trailing_tag_prefix_len(content: &str, tag: &str) -> usize {
3930 let max_len = tag.len().min(content.len());
3931 for len in (1..=max_len).rev() {
3932 let start = content.len() - len;
3933 if content.is_char_boundary(start) && tag.starts_with(&content[start..]) {
3934 return len;
3935 }
3936 }
3937 0
3938 }
3939
3940 fn push_reasoning_segment(segments: &mut Vec<ReasoningSegment>, segment: ReasoningSegment) {
3941 match &segment {
3942 ReasoningSegment::Text(text) | ReasoningSegment::Thinking(text) if text.is_empty() => {}
3943 _ => segments.push(segment),
3944 }
3945 }
3946
3947 fn push_text_delta(
3948 events: &mut Vec<StreamEvent>,
3949 content_index: &mut u32,
3950 text_started: &mut bool,
3951 thinking_started: &mut bool,
3952 text: String,
3953 ) {
3954 if *thinking_started {
3955 events.push(StreamEvent::ContentBlockStop {
3956 index: *content_index,
3957 });
3958 *content_index += 1;
3959 *thinking_started = false;
3960 }
3961 if !*text_started {
3962 events.push(StreamEvent::ContentBlockStart {
3963 index: *content_index,
3964 content_block: ContentBlockStart::Text {
3965 text: String::new(),
3966 },
3967 });
3968 *text_started = true;
3969 }
3970 events.push(StreamEvent::ContentBlockDelta {
3971 index: *content_index,
3972 delta: Delta::TextDelta { text },
3973 });
3974 }
3975
3976 fn push_thinking_delta(
3977 events: &mut Vec<StreamEvent>,
3978 content_index: &mut u32,
3979 text_started: &mut bool,
3980 thinking_started: &mut bool,
3981 thinking: String,
3982 ) {
3983 if *text_started {
3984 events.push(StreamEvent::ContentBlockStop {
3985 index: *content_index,
3986 });
3987 *content_index += 1;
3988 *text_started = false;
3989 }
3990 if !*thinking_started {
3991 events.push(StreamEvent::ContentBlockStart {
3992 index: *content_index,
3993 content_block: ContentBlockStart::Thinking {
3994 thinking: String::new(),
3995 },
3996 });
3997 *thinking_started = true;
3998 }
3999 events.push(StreamEvent::ContentBlockDelta {
4000 index: *content_index,
4001 delta: Delta::ThinkingDelta { thinking },
4002 });
4003 }
4004
4005 // === SSE Chunk Parser ===
4006
4007 enum SseDataFrame {
4008 Done,
4009 Events(Vec<StreamEvent>),
4010 }
4011
4012 // The six `&mut` streaming-state fields plus the style flag are a deliberate,
4013 // shared parser-state set (mirrored by `parse_sse_chunk*`); bundling them into a
4014 // struct would only add reborrow noise on this hot SSE path.
4015 #[allow(clippy::too_many_arguments)]
4016 fn parse_sse_data_frame(
4017 data: &str,
4018 content_index: &mut u32,
4019 text_started: &mut bool,
4020 thinking_started: &mut bool,
4021 tool_indices: &mut std::collections::HashMap<u32, u32>,
4022 reasoning_detail_buffers: &mut std::collections::HashMap<u32, String>,
4023 inline_reasoning_tags: &mut InlineReasoningTagState,
4024 reasoning_stream_style: ReasoningStreamStyle,
4025 ) -> SseDataFrame {
4026 if data.trim() == "[DONE]" {
4027 return SseDataFrame::Done;
4028 }
4029 let events = serde_json::from_str::<Value>(data).map_or_else(
4030 |_| Vec::new(),
4031 |chunk_json| {
4032 parse_sse_chunk_with_reasoning_style(
4033 &chunk_json,
4034 content_index,
4035 text_started,
4036 thinking_started,
4037 tool_indices,
4038 reasoning_detail_buffers,
4039 inline_reasoning_tags,
4040 reasoning_stream_style,
4041 )
4042 },
4043 );
4044 SseDataFrame::Events(events)
4045 }
4046
4047 /// Parse a single SSE chunk from the Chat Completions streaming API into
4048 /// our internal `StreamEvent` representation.
4049 #[cfg(test)]
4050 pub(super) fn parse_sse_chunk(
4051 chunk: &Value,
4052 content_index: &mut u32,
4053 text_started: &mut bool,
4054 thinking_started: &mut bool,
4055 tool_indices: &mut std::collections::HashMap<u32, u32>,
4056 reasoning_detail_buffers: &mut std::collections::HashMap<u32, String>,
4057 is_reasoning_model: bool,
4058 ) -> Vec<StreamEvent> {
4059 let mut inline_reasoning_tags = InlineReasoningTagState::default();
4060 let reasoning_stream_style = if is_reasoning_model {
4061 ReasoningStreamStyle::SeparateField
4062 } else {
4063 ReasoningStreamStyle::None
4064 };
4065 parse_sse_chunk_with_reasoning_style(
4066 chunk,
4067 content_index,
4068 text_started,
4069 thinking_started,
4070 tool_indices,
4071 reasoning_detail_buffers,
4072 &mut inline_reasoning_tags,
4073 reasoning_stream_style,
4074 )
4075 }
4076
4077 // Same deliberate shared parser-state set as `parse_sse_data_frame`.
4078 #[allow(clippy::too_many_arguments)]
4079 fn parse_sse_chunk_with_reasoning_style(
4080 chunk: &Value,
4081 content_index: &mut u32,
4082 text_started: &mut bool,
4083 thinking_started: &mut bool,
4084 tool_indices: &mut std::collections::HashMap<u32, u32>,
4085 reasoning_detail_buffers: &mut std::collections::HashMap<u32, String>,
4086 inline_reasoning_tags: &mut InlineReasoningTagState,
4087 reasoning_stream_style: ReasoningStreamStyle,
4088 ) -> Vec<StreamEvent> {
4089 let mut events = Vec::new();
4090
4091 // OpenAI-compatible providers surface mid-stream failures as a chunk-level
4092 // `error` object (sometimes with `type: "error"`), delivered before
4093 // `[DONE]`. Silently dropping it turned rate-limit / context-length /
4094 // server errors into a truncated turn that looked successful — the frame
4095 // is now surfaced through the same `StreamEvent::Error` contract the
4096 // Anthropic path uses (#3014, ops R3).
4097 if let Some(error) = chunk.get("error") {
4098 let error = match error {
4099 Value::Object(_) => error.clone(),
4100 Value::String(message) => serde_json::json!({ "message": message }),
4101 _ => serde_json::json!({ "message": "provider stream error" }),
4102 };
4103 events.push(StreamEvent::Error { error });
4104 return events;
4105 }
4106
4107 let Some(choices) = chunk.get("choices").and_then(Value::as_array) else {
4108 // Usage-only chunk (sent at end with stream_options)
4109 if let Some(usage_val) = chunk.get("usage") {
4110 let usage = parse_usage(Some(usage_val));
4111 events.push(StreamEvent::MessageDelta {
4112 delta: MessageDelta {
4113 stop_reason: None,
4114 stop_sequence: None,
4115 },
4116 usage: Some(usage),
4117 });
4118 }
4119 return events;
4120 };
4121
4122 if choices.is_empty() {
4123 if let Some(usage_val) = chunk.get("usage") {
4124 let usage = parse_usage(Some(usage_val));
4125 events.push(StreamEvent::MessageDelta {
4126 delta: MessageDelta {
4127 stop_reason: None,
4128 stop_sequence: None,
4129 },
4130 usage: Some(usage),
4131 });
4132 }
4133 return events;
4134 }
4135
4136 for choice in choices {
4137 let choice_index = choice.get("index").and_then(Value::as_u64).unwrap_or(0) as u32;
4138 let delta = choice.get("delta");
4139 let finish_reason = choice
4140 .get("finish_reason")
4141 .and_then(Value::as_str)
4142 .map(str::to_string);
4143
4144 if let Some(delta) = delta {
4145 let reasoning_text = reasoning_delta(delta, choice_index, reasoning_detail_buffers)
4146 .filter(|s| !s.is_empty());
4147 // Mistral la Plateforme streams reasoning as a polymorphic
4148 // `delta.content` value: an array of typed {type: thinking|text}
4149 // blocks while thinking, then a plain string once the final
4150 // answer starts. Flatten thinking sub-blocks into a single
4151 // reasoning delta and treat text sub-blocks as normal content
4152 // before the shared string fallback below.
4153 let (mistral_thinking, mistral_text) =
4154 if reasoning_stream_style == ReasoningStreamStyle::MistralBlocks {
4155 extract_mistral_polymorphic_content(delta)
4156 } else {
4157 (None, None)
4158 };
4159 if let Some(reasoning) = mistral_thinking.as_deref() {
4160 push_thinking_delta(
4161 &mut events,
4162 content_index,
4163 text_started,
4164 thinking_started,
4165 reasoning.to_string(),
4166 );
4167 }
4168 let content_text = mistral_text.or_else(|| {
4169 delta
4170 .get("content")
4171 .and_then(Value::as_str)
4172 .filter(|s| !s.is_empty())
4173 .map(str::to_string)
4174 });
4175
4176 // Handle reasoning_content / reasoning thinking deltas.
4177 if reasoning_stream_style == ReasoningStreamStyle::SeparateField
4178 && let Some(reasoning) = reasoning_text.as_deref()
4179 {
4180 push_thinking_delta(
4181 &mut events,
4182 content_index,
4183 text_started,
4184 thinking_started,
4185 reasoning.to_string(),
4186 );
4187 }
4188
4189 // Generic OpenAI-compatible proxies sometimes stream answer text
4190 // in `reasoning_content`. If this route is configured with no
4191 // reasoning semantics, render that field as normal text when no
4192 // `content` delta is present.
4193 match (content_text, reasoning_stream_style) {
4194 (Some(content), ReasoningStreamStyle::InlineTags) => {
4195 for segment in inline_reasoning_segments(&content, inline_reasoning_tags, false)
4196 {
4197 match segment {
4198 ReasoningSegment::Text(text) => push_text_delta(
4199 &mut events,
4200 content_index,
4201 text_started,
4202 thinking_started,
4203 text,
4204 ),
4205 ReasoningSegment::Thinking(thinking) => push_thinking_delta(
4206 &mut events,
4207 content_index,
4208 text_started,
4209 thinking_started,
4210 thinking,
4211 ),
4212 }
4213 }
4214 }
4215 (Some(content), _) => push_text_delta(
4216 &mut events,
4217 content_index,
4218 text_started,
4219 thinking_started,
4220 content,
4221 ),
4222 (None, ReasoningStreamStyle::None) => {
4223 if let Some(content) = reasoning_text {
4224 push_text_delta(
4225 &mut events,
4226 content_index,
4227 text_started,
4228 thinking_started,
4229 content,
4230 );
4231 }
4232 }
4233 (None, _) => {}
4234 }
4235
4236 // Handle tool calls
4237 if let Some(tool_calls) = delta.get("tool_calls").and_then(Value::as_array) {
4238 for tc in tool_calls {
4239 let tc_index = tc.get("index").and_then(Value::as_u64).unwrap_or(0) as u32;
4240 let tool_block_index = match tool_indices.entry(tc_index) {
4241 std::collections::hash_map::Entry::Occupied(entry) => *entry.get(),
4242 std::collections::hash_map::Entry::Vacant(entry) => {
4243 // Close text block if transitioning to tool use
4244 if *text_started {
4245 events.push(StreamEvent::ContentBlockStop {
4246 index: *content_index,
4247 });
4248 *content_index += 1;
4249 *text_started = false;
4250 }
4251 if *thinking_started {
4252 events.push(StreamEvent::ContentBlockStop {
4253 index: *content_index,
4254 });
4255 *content_index += 1;
4256 *thinking_started = false;
4257 }
4258
4259 let block_index = *content_index;
4260 let id = tc
4261 .get("id")
4262 .and_then(Value::as_str)
4263 .map(str::to_string)
4264 // Some upstream gateways (and the responses-API
4265 // bridge) elide the `id` on the first chunk of a
4266 // tool call. Falling back to a constant string
4267 // collides when the model emits parallel tool
4268 // calls in the same delta — every call ended up
4269 // with the same id and downstream tool-result
4270 // routing matched the first one twice. Index by
4271 // the content-block position to keep the
4272 // fallback unique within the response.
4273 .unwrap_or_else(|| format!("call_{block_index}"));
4274 let name = tc
4275 .get("function")
4276 .and_then(|f| f.get("name"))
4277 .and_then(Value::as_str);
4278 let name = tool_name_or_fallback(name, &id, "Streaming response chunk");
4279 let caller = tc.get("caller").and_then(|v| {
4280 v.get("type").and_then(Value::as_str).map(|caller_type| {
4281 ToolCaller {
4282 caller_type: caller_type.to_string(),
4283 tool_id: v
4284 .get("tool_id")
4285 .and_then(Value::as_str)
4286 .map(std::string::ToString::to_string),
4287 }
4288 })
4289 });
4290
4291 let thought_signature = tc
4292 .pointer("/extra_content/google/thought_signature")
4293 .and_then(Value::as_str)
4294 .map(str::to_string);
4295 events.push(StreamEvent::ContentBlockStart {
4296 index: block_index,
4297 content_block: ContentBlockStart::ToolUse {
4298 id,
4299 name: from_api_tool_name(&name),
4300 input: json!({}),
4301 caller,
4302 thought_signature,
4303 },
4304 });
4305 *content_index = (*content_index).saturating_add(1);
4306 entry.insert(block_index);
4307 block_index
4308 }
4309 };
4310
4311 // Stream tool call arguments
4312 if let Some(args) = tc
4313 .get("function")
4314 .and_then(|f| f.get("arguments"))
4315 .and_then(Value::as_str)
4316 && !args.is_empty()
4317 {
4318 events.push(StreamEvent::ContentBlockDelta {
4319 index: tool_block_index,
4320 delta: Delta::InputJsonDelta {
4321 partial_json: args.to_string(),
4322 },
4323 });
4324 }
4325 }
4326 }
4327 }
4328
4329 // Handle finish reason
4330 if let Some(reason) = finish_reason {
4331 if reasoning_stream_style == ReasoningStreamStyle::InlineTags {
4332 for segment in inline_reasoning_segments("", inline_reasoning_tags, true) {
4333 match segment {
4334 ReasoningSegment::Text(text) => push_text_delta(
4335 &mut events,
4336 content_index,
4337 text_started,
4338 thinking_started,
4339 text,
4340 ),
4341 ReasoningSegment::Thinking(thinking) => push_thinking_delta(
4342 &mut events,
4343 content_index,
4344 text_started,
4345 thinking_started,
4346 thinking,
4347 ),
4348 }
4349 }
4350 }
4351 // Close any open blocks
4352 if *text_started {
4353 events.push(StreamEvent::ContentBlockStop {
4354 index: *content_index,
4355 });
4356 *text_started = false;
4357 }
4358 if *thinking_started {
4359 events.push(StreamEvent::ContentBlockStop {
4360 index: *content_index,
4361 });
4362 *thinking_started = false;
4363 }
4364 // Close tool blocks
4365 let mut open_tool_indices: Vec<u32> =
4366 tool_indices.drain().map(|(_, idx)| idx).collect();
4367 open_tool_indices.sort_unstable();
4368 for tool_block_index in open_tool_indices {
4369 events.push(StreamEvent::ContentBlockStop {
4370 index: tool_block_index,
4371 });
4372 }
4373
4374 // Emit usage from the chunk if available
4375 let chunk_usage = chunk.get("usage").map(|u| parse_usage(Some(u)));
4376 events.push(StreamEvent::MessageDelta {
4377 delta: MessageDelta {
4378 stop_reason: Some(reason),
4379 stop_sequence: None,
4380 },
4381 usage: chunk_usage,
4382 });
4383 }
4384 }
4385
4386 events
4387 }
4388
4389 fn tool_name_or_fallback(name: Option<&str>, id: &str, source: &str) -> String {
4390 let trimmed = name.unwrap_or("").trim();
4391 if trimmed.is_empty() {
4392 logging::warn(format!(
4393 "{source} returned an empty tool name for call {id}; using unknown_tool"
4394 ));
4395 "unknown_tool".to_string()
4396 } else {
4397 trimmed.to_string()
4398 }
4399 }
4400
4401 // === #103 Phase 1: stream-decode diagnostics ===================================
4402
4403 #[cfg(test)]
4404 mod stream_diagnostics_tests {
4405 use super::*;
4406 use reqwest::header::{HeaderMap, HeaderValue};
4407
4408 #[test]
4409 fn stream_idle_timeout_reports_progress_and_timing() {
4410 let message = super::super::stream_entry::idle_timeout_message(
4411 Duration::from_secs(240),
4412 8192,
4413 Duration::from_millis(73_500),
4414 Duration::from_millis(41_250),
4415 );
4416
4417 assert_eq!(
4418 message,
4419 "SSE stream idle timeout after 240s — no data received \
4420 (bytes_received=8192, stream_age_ms=73500, ms_since_last_chunk=41250)"
4421 );
4422 }
4423
4424 #[test]
4425 fn chat_completions_error_frames_surface_as_stream_events() {
4426 let mut content_index = 0u32;
4427 let mut text_started = false;
4428 let mut thinking_started = false;
4429 let mut tool_indices = std::collections::HashMap::new();
4430 let mut reasoning_buffers = std::collections::HashMap::new();
4431 for chunk in [
4432 json!({ "error": { "message": "rate limit exceeded", "type": "rate_limit_error" } }),
4433 json!({ "type": "error", "error": { "message": "context length exceeded" } }),
4434 json!({ "error": "server error" }),
4435 ] {
4436 let events = parse_sse_chunk(
4437 &chunk,
4438 &mut content_index,
4439 &mut text_started,
4440 &mut thinking_started,
4441 &mut tool_indices,
4442 &mut reasoning_buffers,
4443 false,
4444 );
4445 assert_eq!(
4446 events.len(),
4447 1,
4448 "a chunk-level error frame must not be swallowed ({chunk})"
4449 );
4450 match &events[0] {
4451 StreamEvent::Error { error } => assert!(
4452 !error.is_null()
4453 && (error.get("message").and_then(Value::as_str).is_some()
4454 || error.is_string()),
4455 "the provider error message must survive parsing: {error}"
4456 ),
4457 other => panic!("expected StreamEvent::Error, got {other:?}"),
4458 }
4459 }
4460 // A normal content chunk still parses as a delta after the error path.
4461 let deltas = parse_sse_chunk(
4462 &json!({"choices": [{"index": 0, "delta": {"content": "ok"}}]}),
4463 &mut content_index,
4464 &mut text_started,
4465 &mut thinking_started,
4466 &mut tool_indices,
4467 &mut reasoning_buffers,
4468 false,
4469 );
4470 assert!(
4471 deltas.iter().any(|event| {
4472 matches!(
4473 event,
4474 StreamEvent::ContentBlockDelta {
4475 delta: Delta::TextDelta { text }, ..
4476 } if text == "ok"
4477 )
4478 }),
4479 "content deltas still parse: {deltas:?}"
4480 );
4481 }
4482
4483 #[test]
4484 fn deepseek_thinking_omits_tool_choice() {
4485 for effort in [Some("high"), Some("max"), Some("medium"), Some("")] {
4486 assert!(
4487 !should_send_tool_choice_for_chat(ProviderKind::Deepseek, effort),
4488 "DeepSeek thinking rejects explicit tool_choice for {effort:?}"
4489 );
4490 assert!(
4491 !should_send_tool_choice_for_chat(ProviderKind::Deepseek, effort),
4492 "DeepSeek CN thinking rejects explicit tool_choice for {effort:?}"
4493 );
4494 }
4495
4496 for effort in [
4497 None,
4498 Some("off"),
4499 Some("disabled"),
4500 Some("none"),
4501 Some("false"),
4502 ] {
4503 assert!(should_send_tool_choice_for_chat(
4504 ProviderKind::Deepseek,
4505 effort
4506 ));
4507 }
4508 assert!(should_send_tool_choice_for_chat(
4509 ProviderKind::Openrouter,
4510 Some("high")
4511 ));
4512 }
4513
4514 #[test]
4515 fn format_stream_headers_renders_all_fields_when_present() {
4516 let mut headers = HeaderMap::new();
4517 headers.insert("content-encoding", HeaderValue::from_static("gzip"));
4518 headers.insert("transfer-encoding", HeaderValue::from_static("chunked"));
4519 headers.insert("connection", HeaderValue::from_static("keep-alive"));
4520 headers.insert("server", HeaderValue::from_static("openresty/1.25.3.1"));
4521
4522 let rendered = format_stream_headers(&headers);
4523 // Order is fixed by FIELDS in the helper; assert each field appears.
4524 assert!(
4525 rendered.contains("content-encoding=gzip"),
4526 "got: {rendered}"
4527 );
4528 assert!(
4529 rendered.contains("transfer-encoding=chunked"),
4530 "got: {rendered}"
4531 );
4532 assert!(
4533 rendered.contains("connection=keep-alive"),
4534 "got: {rendered}"
4535 );
4536 assert!(
4537 rendered.contains("server=openresty/1.25.3.1"),
4538 "got: {rendered}"
4539 );
4540 }
4541
4542 #[test]
4543 fn format_stream_headers_marks_missing_fields_as_absent() {
4544 // DeepSeek frequently omits content-encoding when not compressing.
4545 // The diagnostic must still produce a parseable line so log scrapers
4546 // don't lose the slot.
4547 let headers = HeaderMap::new();
4548 let rendered = format_stream_headers(&headers);
4549 assert!(
4550 rendered.contains("content-encoding=(absent)"),
4551 "missing field must be explicitly marked; got: {rendered}"
4552 );
4553 assert!(
4554 rendered.contains("transfer-encoding=(absent)"),
4555 "missing field must be explicitly marked; got: {rendered}"
4556 );
4557 }
4558
4559 #[test]
4560 fn format_stream_headers_handles_non_ascii_value_gracefully() {
4561 // If a header value isn't UTF-8, `.to_str()` fails — we must not panic
4562 // and should still produce a parseable line.
4563 let mut headers = HeaderMap::new();
4564 // 0xFF is a valid byte but invalid UTF-8 start byte.
4565 headers.insert(
4566 "server",
4567 HeaderValue::from_bytes(b"\xff\xfemystery").expect("header value"),
4568 );
4569 let rendered = format_stream_headers(&headers);
4570 assert!(
4571 rendered.contains("server=(absent)"),
4572 "non-UTF8 header values fall back to (absent); got: {rendered}"
4573 );
4574 }
4575 }
4576
4577 #[cfg(test)]
4578 mod arcee_waf_message_encoding_tests {
4579 use super::build_chat_messages_for_request_and_provider;
4580 use crate::config::ProviderKind;
4581 use codewhale_models::{MessageRequest, SystemPrompt};
4582 use serde_json::Value;
4583
4584 fn request_with_system(system: &str) -> MessageRequest {
4585 MessageRequest {
4586 model: "trinity-large-thinking".to_string(),
4587 messages: Vec::new(),
4588 max_tokens: 16,
4589 system: Some(SystemPrompt::Text(system.to_string())),
4590 tools: None,
4591 tool_choice: None,
4592 metadata: None,
4593 thinking: None,
4594 reasoning_effort: None,
4595 stream: None,
4596 temperature: None,
4597 top_p: None,
4598 }
4599 }
4600
4601 fn decoded_content(content: &Value) -> String {
4602 if let Some(text) = content.as_str() {
4603 return text.to_string();
4604 }
4605 content
4606 .as_array()
4607 .expect("content parts")
4608 .iter()
4609 .map(|part| part.get("text").and_then(Value::as_str).expect("text part"))
4610 .collect()
4611 }
4612
4613 #[test]
4614 fn arcee_splits_waf_trigger_without_changing_decoded_system_prompt() {
4615 let system = "Run calculations with `python -c 'print(1)'` when a tool is available.";
4616 let request = request_with_system(system);
4617
4618 let messages = build_chat_messages_for_request_and_provider(&request, ProviderKind::Arcee);
4619 let content = &messages[0]["content"];
4620
4621 assert!(
4622 content.is_array(),
4623 "Arcee system content with a WAF trigger should be encoded as text parts"
4624 );
4625 assert_eq!(decoded_content(content), system);
4626 let serialized = serde_json::to_string(&messages).expect("serialize messages");
4627 assert!(
4628 !serialized.contains("python -c"),
4629 "wire JSON should not contain the Cloudflare trigger contiguously: {serialized}"
4630 );
4631 }
4632
4633 #[test]
4634 fn non_arcee_providers_keep_system_prompt_as_string() {
4635 let system = "Run calculations with `python -c 'print(1)'` when a tool is available.";
4636 let request = request_with_system(system);
4637
4638 let messages = build_chat_messages_for_request_and_provider(&request, ProviderKind::Openai);
4639
4640 assert_eq!(messages[0]["content"].as_str(), Some(system));
4641 }
4642
4643 #[test]
4644 fn arcee_keeps_non_triggering_system_prompt_as_string() {
4645 let system = "Use read-only tools to inspect files before reporting results.";
4646 let request = request_with_system(system);
4647
4648 let messages = build_chat_messages_for_request_and_provider(&request, ProviderKind::Arcee);
4649
4650 assert_eq!(messages[0]["content"].as_str(), Some(system));
4651 }
4652 }
4653
4654 #[cfg(test)]
4655 mod minimax_reasoning_replay_tests {
4656 use super::{
4657 build_chat_messages_for_request_and_provider,
4658 build_chat_messages_for_request_and_provider_and_route,
4659 };
4660 use crate::config::{
4661 DEFAULT_KIMI_CODE_BASE_URL, DEFAULT_MINIMAX_MODEL, DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL,
4662 DEFAULT_MOONSHOT_BASE_URL, KIMI_CODE_K3_MODEL, ProviderKind,
4663 };
4664 use codewhale_models::Role;
4665 use codewhale_models::{ContentBlock, Message, MessageRequest};
4666
4667 fn request_with_assistant_thinking() -> MessageRequest {
4668 MessageRequest {
4669 model: DEFAULT_MINIMAX_MODEL.to_string(),
4670 messages: vec![Message {
4671 role: Role::Assistant,
4672 content: vec![
4673 ContentBlock::Thinking {
4674 thinking: "Inspect tool state".to_string(),
4675 signature: None,
4676 state: None,
4677 },
4678 ContentBlock::Text {
4679 text: "Done.".to_string(),
4680 cache_control: None,
4681 },
4682 ],
4683 }],
4684 max_tokens: 16,
4685 system: None,
4686 tools: None,
4687 tool_choice: None,
4688 metadata: None,
4689 thinking: None,
4690 reasoning_effort: None,
4691 stream: None,
4692 temperature: None,
4693 top_p: None,
4694 }
4695 }
4696
4697 #[test]
4698 fn minimax_history_replays_thinking_as_reasoning_details() {
4699 let request = request_with_assistant_thinking();
4700
4701 let messages =
4702 build_chat_messages_for_request_and_provider(&request, ProviderKind::Minimax);
4703 let assistant = &messages[0];
4704
4705 assert_eq!(
4706 assistant
4707 .get("reasoning_content")
4708 .and_then(|value| value.as_str()),
4709 Some("Inspect tool state")
4710 );
4711 assert_eq!(
4712 assistant
4713 .pointer("/reasoning_details/0/type")
4714 .and_then(|value| value.as_str()),
4715 Some("text")
4716 );
4717 assert_eq!(
4718 assistant
4719 .pointer("/reasoning_details/0/text")
4720 .and_then(|value| value.as_str()),
4721 Some("Inspect tool state")
4722 );
4723 }
4724
4725 #[test]
4726 fn kimi_code_k3_replays_thinking_only_on_the_exact_membership_route() {
4727 let mut request = request_with_assistant_thinking();
4728 request.model = KIMI_CODE_K3_MODEL.to_string();
4729
4730 let exact = build_chat_messages_for_request_and_provider_and_route(
4731 &request,
4732 ProviderKind::Moonshot,
4733 DEFAULT_KIMI_CODE_BASE_URL,
4734 );
4735 assert_eq!(
4736 exact[0]
4737 .get("reasoning_content")
4738 .and_then(serde_json::Value::as_str),
4739 Some("Inspect tool state")
4740 );
4741
4742 let neighbor = build_chat_messages_for_request_and_provider_and_route(
4743 &request,
4744 ProviderKind::Moonshot,
4745 DEFAULT_MOONSHOT_BASE_URL,
4746 );
4747 assert!(
4748 neighbor[0].get("reasoning_content").is_none(),
4749 "a generic Moonshot k3 identifier must not inherit Kimi Code replay"
4750 );
4751 }
4752
4753 #[test]
4754 fn modelstudio_qwen38_wire_body_replays_no_historical_reasoning_across_tool_loop() {
4755 // One user handoff, one assistant Thinking + Text + ToolUse turn, and
4756 // its matching ToolResult. On the exact Model Studio route the
4757 // always-thinking qwen3.8 family must not receive historical
4758 // `reasoning_content`, while the handoff occurs exactly once and the
4759 // tool call id, arguments, and result stay intact.
4760 let mut request = request_with_assistant_thinking();
4761 request.model = "qwen3.8-max".to_string();
4762 request.messages = vec![
4763 Message {
4764 role: Role::User,
4765 content: vec![ContentBlock::Text {
4766 text: "HANDOFF-SENTINEL: fix the widget.".to_string(),
4767 cache_control: None,
4768 }],
4769 },
4770 Message {
4771 role: Role::Assistant,
4772 content: vec![
4773 ContentBlock::Thinking {
4774 thinking: "stale thinking from the prior turn".to_string(),
4775 signature: None,
4776 state: None,
4777 },
4778 ContentBlock::Text {
4779 text: "I'll read the widget first.".to_string(),
4780 cache_control: None,
4781 },
4782 ContentBlock::ToolUse {
4783 execution_id: None,
4784 id: "call_qwen38_001".to_string(),
4785 name: "read".to_string(),
4786 input: serde_json::json!({ "path": "widget.rs" }),
4787 caller: None,
4788 thought_signature: None,
4789 },
4790 ],
4791 },
4792 Message {
4793 role: Role::User,
4794 content: vec![ContentBlock::ToolResult {
4795 execution_id: None,
4796 tool_use_id: "call_qwen38_001".to_string(),
4797 content: "widget.rs: struct Widget { .. }".to_string(),
4798 is_error: None,
4799 content_blocks: None,
4800 }],
4801 },
4802 ];
4803
4804 let messages = build_chat_messages_for_request_and_provider_and_route(
4805 &request,
4806 ProviderKind::ModelstudioTokenPlan,
4807 DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL,
4808 );
4809
4810 // The handoff occurs exactly once, as a plain user text message.
4811 let handoff_carriers: Vec<&serde_json::Value> = messages
4812 .iter()
4813 .filter(|message| {
4814 message.get("role").and_then(serde_json::Value::as_str) == Some("user")
4815 && message
4816 .get("content")
4817 .and_then(serde_json::Value::as_str)
4818 .is_some_and(|content| content.contains("HANDOFF-SENTINEL"))
4819 })
4820 .collect();
4821 assert_eq!(
4822 handoff_carriers.len(),
4823 1,
4824 "the handoff must appear in exactly one user message: {messages:?}"
4825 );
4826
4827 // The assistant turn keeps its text and tool call, but no historical
4828 // reasoning_content for qwen3.8.
4829 let assistant = messages
4830 .iter()
4831 .find(|message| {
4832 message.get("role").and_then(serde_json::Value::as_str) == Some("assistant")
4833 })
4834 .expect("assistant message");
4835 assert_eq!(
4836 assistant.get("content").and_then(serde_json::Value::as_str),
4837 Some("I'll read the widget first.")
4838 );
4839 assert!(
4840 assistant.get("reasoning_content").is_none(),
4841 "qwen3.8 must not receive historical reasoning_content: {assistant:?}"
4842 );
4843 let tool_calls = assistant
4844 .get("tool_calls")
4845 .and_then(serde_json::Value::as_array)
4846 .expect("tool_calls array");
4847 assert_eq!(tool_calls.len(), 1);
4848 assert_eq!(tool_calls[0]["id"], serde_json::json!("call_qwen38_001"));
4849 assert_eq!(
4850 tool_calls[0].pointer("/function/name"),
4851 Some(&serde_json::json!("read"))
4852 );
4853 assert_eq!(
4854 tool_calls[0].pointer("/function/arguments"),
4855 Some(&serde_json::json!(r#"{"path":"widget.rs"}"#))
4856 );
4857
4858 // The matching tool result rides along under its original id.
4859 let tool_result = messages
4860 .iter()
4861 .find(|message| message.get("role").and_then(serde_json::Value::as_str) == Some("tool"))
4862 .expect("tool result message");
4863 assert_eq!(
4864 tool_result.get("tool_call_id"),
4865 Some(&serde_json::json!("call_qwen38_001"))
4866 );
4867 assert_eq!(
4868 tool_result
4869 .get("content")
4870 .and_then(serde_json::Value::as_str),
4871 Some("widget.rs: struct Widget { .. }")
4872 );
4873
4874 // Control: the same handoff/tool loop with a documented
4875 // preserve-thinking model replays its historical reasoning, proving
4876 // the strip above is model-gated, not route-gated.
4877 let mut preserve_request = request;
4878 preserve_request.model = "qwen3.7-plus".to_string();
4879 let preserve_messages = build_chat_messages_for_request_and_provider_and_route(
4880 &preserve_request,
4881 ProviderKind::ModelstudioTokenPlan,
4882 DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL,
4883 );
4884 let preserve_assistant = preserve_messages
4885 .iter()
4886 .find(|message| {
4887 message.get("role").and_then(serde_json::Value::as_str) == Some("assistant")
4888 })
4889 .expect("assistant message");
4890 assert_eq!(
4891 preserve_assistant
4892 .get("reasoning_content")
4893 .and_then(serde_json::Value::as_str),
4894 Some("stale thinking from the prior turn"),
4895 "documented preserve-thinking models keep replaying history"
4896 );
4897 }
4898 }
4899
4900 // === #103 Phase 4: SSE decoder behavior on canned chunk sequences ============
4901
4902 #[cfg(test)]
4903 #[path = "chat/tests/stream_decoder.rs"]
4904 mod stream_decoder_tests;
4905
4906 #[cfg(test)]
4907 mod alias_thinking_detection_tests {
4908 //! Regression coverage for the DeepSeek public model aliases.
4909 //!
4910 //! `deepseek-chat` and `deepseek-reasoner` are the canonical alias names
4911 //! published in DeepSeek's API docs. Server-side they resolve to V4-flash
4912 //! and V4-pro respectively, both of which have thinking mode enabled by
4913 //! default. If the TUI does not classify those aliases as reasoning
4914 //! models, the sanitizer skips replaying `reasoning_content` on tool-call
4915 //! assistant messages and DeepSeek returns a 400 ("the `reasoning_content`
4916 //! in the thinking mode must be passed back to the API") on the second
4917 //! turn. See upstream API docs:
4918 //! <https://api-docs.deepseek.com/guides/thinking_mode>
4919 use super::{
4920 ReasoningStreamStyle, apply_direct_moonshot_k3_fixed_sampling,
4921 apply_inkling_reasoning_effort, apply_kimi_code_fixed_sampling,
4922 apply_kimi_code_k3_reasoning_effort, apply_openai_reasoning_effort,
4923 apply_provider_token_limit, apply_route_reasoning_controls, is_reasoning_model_for_stream,
4924 provider_accepts_reasoning_content, reasoning_stream_style_for_route,
4925 requires_reasoning_content, should_replay_reasoning_content,
4926 should_replay_reasoning_content_for_provider,
4927 should_replay_reasoning_content_for_provider_on_route,
4928 };
4929 use crate::config::ProviderKind;
4930 use serde_json::json;
4931
4932 #[test]
4933 fn aliases_routed_to_v4_require_reasoning_content() {
4934 // Documented public aliases.
4935 assert!(requires_reasoning_content("deepseek-chat"));
4936 assert!(requires_reasoning_content("deepseek-reasoner"));
4937 // Case-insensitive: users sometimes copy/paste with capitalisation.
4938 assert!(requires_reasoning_content("DeepSeek-Chat"));
4939 assert!(requires_reasoning_content("DEEPSEEK-REASONER"));
4940 }
4941
4942 #[test]
4943 fn explicit_v4_ids_still_require_reasoning_content() {
4944 // Direct V4 IDs continue to match (regression guard for the existing
4945 // `lower.contains("deepseek-v4")` branch).
4946 assert!(requires_reasoning_content("deepseek-v4-flash"));
4947 assert!(requires_reasoning_content("deepseek-v4-pro"));
4948 }
4949
4950 #[test]
4951 fn non_thinking_aliases_remain_excluded() {
4952 // Legacy non-thinking IDs and unrelated provider models must not be
4953 // misclassified, otherwise we would force a placeholder
4954 // `reasoning_content` on providers that reject the field.
4955 assert!(!requires_reasoning_content("deepseek-v3"));
4956 assert!(!requires_reasoning_content("deepseek-coder"));
4957 assert!(!requires_reasoning_content("qwen3-coder"));
4958 assert!(!requires_reasoning_content("claude-sonnet-4-6"));
4959 }
4960
4961 #[test]
4962 fn alias_prefix_handles_suffixed_variants() {
4963 // OpenRouter / proxy deployments occasionally suffix the canonical
4964 // alias (e.g. `deepseek-chat:free`). Those routes still hit V4
4965 // server-side, so they must continue to require reasoning_content.
4966 assert!(requires_reasoning_content("deepseek-chat:free"));
4967 assert!(requires_reasoning_content("deepseek-reasoner-2025-05"));
4968 }
4969
4970 #[test]
4971 fn explicit_reasoning_off_overrides_alias_detection() {
4972 // `reasoning_effort = "off"` is the documented escape hatch: even when
4973 // the model is in the thinking family, the user can opt out and the
4974 // sanitizer must respect that choice.
4975 assert!(!should_replay_reasoning_content(
4976 "deepseek-chat",
4977 Some("off")
4978 ));
4979 assert!(!should_replay_reasoning_content(
4980 "deepseek-reasoner",
4981 Some("disabled")
4982 ));
4983 // Without an explicit override, alias models still trigger replay.
4984 assert!(should_replay_reasoning_content("deepseek-chat", None));
4985 assert!(should_replay_reasoning_content(
4986 "deepseek-reasoner",
4987 Some("medium")
4988 ));
4989 }
4990
4991 #[test]
4992 fn generic_openai_provider_does_not_accept_reasoning_content_semantics() {
4993 assert!(!provider_accepts_reasoning_content(ProviderKind::Openai));
4994 assert!(provider_accepts_reasoning_content(ProviderKind::Deepseek));
4995 assert!(provider_accepts_reasoning_content(ProviderKind::NvidiaNim));
4996 assert!(provider_accepts_reasoning_content(ProviderKind::XiaomiMimo));
4997 assert!(provider_accepts_reasoning_content(ProviderKind::Arcee));
4998 assert!(provider_accepts_reasoning_content(ProviderKind::Minimax));
4999 assert!(provider_accepts_reasoning_content(ProviderKind::Zai));
5000 // #3016: Moonshot's native endpoint streams Kimi thinking as
5001 // reasoning_content.
5002 assert!(provider_accepts_reasoning_content(ProviderKind::Moonshot));
5003 }
5004
5005 /// Alibaba's classic pay-as-you-go DashScope endpoints are genuine
5006 /// Alibaba Chat Completions hosts serving the same models; before
5007 /// 2026-08-04 they were missing from the verifier allowlist, so every
5008 /// reasoning control was silently stripped there (fail-closed feature
5009 /// loss, not a leak). The intl spelling matches provider_defaults.
5010 #[test]
5011 fn classic_dashscope_hosts_are_verified_modelstudio_chat_routes() {
5012 for base_url in [
5013 "https://dashscope.aliyuncs.com/compatible-mode/v1",
5014 "https://dashscope-intl.aliyuncs.com/compatible-mode/v1",
5015 "https://dashscope-intl.aliyuncs.com/compatible-mode/v1/",
5016 ] {
5017 assert!(
5018 super::is_exact_modelstudio_chat_route(
5019 ProviderKind::ModelstudioTokenPlan,
5020 base_url
5021 ),
5022 "{base_url}"
5023 );
5024 let mut body = json!({});
5025 apply_route_reasoning_controls(
5026 &mut body,
5027 ProviderKind::ModelstudioTokenPlan,
5028 base_url,
5029 "qwen3.7-plus",
5030 Some("off"),
5031 );
5032 assert_eq!(body["enable_thinking"], json!(false), "{base_url}: {body}");
5033 }
5034 // Lookalike hosts stay unverified — fail closed.
5035 for base_url in [
5036 "https://dashscope.aliyuncs.com.evil.example/compatible-mode/v1",
5037 "https://notdashscope.aliyuncs.com/compatible-mode/v1",
5038 "https://dashscope.aliyuncs.com/other-path/v1",
5039 ] {
5040 assert!(
5041 !super::is_exact_modelstudio_chat_route(
5042 ProviderKind::ModelstudioTokenPlan,
5043 base_url
5044 ),
5045 "{base_url}"
5046 );
5047 }
5048 }
5049
5050 #[test]
5051 fn modelstudio_hybrid_routes_send_documented_thinking_controls() {
5052 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5053 for (effort, enabled) in [
5054 (None, true),
5055 (Some("low"), true),
5056 (Some("high"), true),
5057 (Some("xhigh"), true),
5058 (Some("off"), false),
5059 ] {
5060 let mut body = json!({});
5061 apply_route_reasoning_controls(
5062 &mut body,
5063 ProviderKind::ModelstudioTokenPlan,
5064 base_url,
5065 "qwen3.7-plus",
5066 effort,
5067 );
5068
5069 assert_eq!(body["enable_thinking"], json!(enabled), "{effort:?}");
5070 assert_eq!(body["preserve_thinking"], json!(enabled), "{effort:?}");
5071 assert!(body.get("thinking").is_none(), "{effort:?}: {body}");
5072 assert!(body.get("reasoning_effort").is_none(), "{effort:?}: {body}");
5073 }
5074 }
5075
5076 #[test]
5077 fn modelstudio_deepseek_v4_maps_effort_to_documented_values() {
5078 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5079 for (requested, expected) in [("low", "high"), ("high", "high"), ("xhigh", "max")] {
5080 let mut body = json!({});
5081 apply_route_reasoning_controls(
5082 &mut body,
5083 ProviderKind::ModelstudioTokenPlan,
5084 base_url,
5085 "deepseek-v4-pro",
5086 Some(requested),
5087 );
5088
5089 assert_eq!(body["enable_thinking"], json!(true), "{requested}");
5090 assert_eq!(body["reasoning_effort"], json!(expected), "{requested}");
5091 }
5092 }
5093
5094 #[test]
5095 fn modelstudio_reasoning_controls_fail_closed_on_custom_gateways() {
5096 let mut body = json!({
5097 "enable_thinking": true,
5098 "preserve_thinking": true,
5099 "reasoning_effort": "high",
5100 });
5101 apply_route_reasoning_controls(
5102 &mut body,
5103 ProviderKind::ModelstudioTokenPlan,
5104 "https://proxy.example/v1",
5105 "qwen3.7-plus",
5106 Some("high"),
5107 );
5108
5109 assert!(body.get("enable_thinking").is_none());
5110 assert!(body.get("preserve_thinking").is_none());
5111 assert!(body.get("reasoning_effort").is_none());
5112 }
5113
5114 #[test]
5115 fn modelstudio_anthropic_identities_write_nothing_on_the_chat_path() {
5116 // The Messages adapter owns these two. If `wire = "openai"` ever routes
5117 // them through Chat Completions, the shaper must strip rather than
5118 // inherit the OpenAI-dialect fields — there is no provider-enum writer
5119 // left to re-add them.
5120 for provider in [
5121 ProviderKind::ModelstudioTokenPlanAnthropic,
5122 ProviderKind::ModelstudioCodingPlanAnthropic,
5123 ] {
5124 for base_url in [
5125 crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL,
5126 crate::config::MODELSTUDIO_TOKEN_PLAN_ANTHROPIC_BASE_URL,
5127 ] {
5128 let mut body = json!({ "enable_thinking": true });
5129 apply_route_reasoning_controls(
5130 &mut body,
5131 provider,
5132 base_url,
5133 "qwen3.7-plus",
5134 Some("high"),
5135 );
5136 assert_eq!(body, json!({}), "{provider:?} {base_url}");
5137 }
5138 }
5139 }
5140
5141 #[test]
5142 fn modelstudio_qwen38_route_streams_reasoning_without_replaying_history() {
5143 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5144 for model in ["qwen3.8-max", "qwen3.8-max-preview"] {
5145 // qwen3.8 is thinking-only. Effort selection must never hide its
5146 // separate reasoning stream, including the stale `off` state
5147 // that can arrive before route normalization.
5148 assert_eq!(
5149 reasoning_stream_style_for_route(
5150 ProviderKind::ModelstudioTokenPlan,
5151 base_url,
5152 model,
5153 None,
5154 ),
5155 ReasoningStreamStyle::SeparateField,
5156 "{model}"
5157 );
5158 // ...but Alibaba does not document `preserve_thinking` for the
5159 // qwen3.8 family, so no historical `reasoning_content` may be
5160 // replayed — even when a stale effort claims thinking is off.
5161 // Replaying it feeds the model its own past Thinking blocks and
5162 // re-triggers them every turn (observed handoff loop).
5163 for effort in [None, Some("off"), Some("high"), Some("xhigh")] {
5164 assert!(
5165 !should_replay_reasoning_content_for_provider_on_route(
5166 ProviderKind::ModelstudioTokenPlan,
5167 base_url,
5168 model,
5169 effort,
5170 ),
5171 "{model} {effort:?}"
5172 );
5173 }
5174 // ...and no enable/disable or preserve switch is ever sent for
5175 // them.
5176 for effort in [None, Some("off"), Some("high")] {
5177 let mut body = json!({});
5178 apply_route_reasoning_controls(
5179 &mut body,
5180 ProviderKind::ModelstudioTokenPlan,
5181 base_url,
5182 model,
5183 effort,
5184 );
5185 assert!(body.get("enable_thinking").is_none(), "{model}: {body}");
5186 assert!(
5187 body.get("preserve_thinking").is_none(),
5188 "{model} {effort:?}: {body}"
5189 );
5190 assert!(body.get("reasoning_effort").is_none(), "{model}: {body}");
5191 }
5192 }
5193 }
5194
5195 #[test]
5196 fn modelstudio_hybrid_route_classifies_reasoning_and_replays_history() {
5197 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5198 assert_eq!(
5199 reasoning_stream_style_for_route(
5200 ProviderKind::ModelstudioTokenPlan,
5201 base_url,
5202 "qwen3.7-plus",
5203 None,
5204 ),
5205 ReasoningStreamStyle::SeparateField,
5206 );
5207 assert!(should_replay_reasoning_content_for_provider_on_route(
5208 ProviderKind::ModelstudioTokenPlan,
5209 base_url,
5210 "qwen3.7-plus",
5211 None,
5212 ));
5213 assert!(!should_replay_reasoning_content_for_provider_on_route(
5214 ProviderKind::ModelstudioTokenPlan,
5215 base_url,
5216 "qwen3.7-plus",
5217 Some("off"),
5218 ));
5219 }
5220
5221 #[test]
5222 fn modelstudio_replay_stays_narrow_until_a_live_key_confirms_it() {
5223 // Deliberately narrower than PR #5233: only `preserve_thinking` models
5224 // replay. GLM and DeepSeek-V3.x on Model Studio stay stripped until
5225 // someone with a key confirms DashScope accepts `reasoning_content` in
5226 // input messages. deepseek-v4* is unaffected — it replays through
5227 // `requires_reasoning_content` on every provider.
5228 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5229 for model in ["glm-5.2", "deepseek-v3.2", "deepseek-v3.1"] {
5230 assert!(
5231 !should_replay_reasoning_content_for_provider_on_route(
5232 ProviderKind::ModelstudioTokenPlan,
5233 base_url,
5234 model,
5235 None,
5236 ),
5237 "{model}"
5238 );
5239 }
5240 assert!(should_replay_reasoning_content_for_provider_on_route(
5241 ProviderKind::ModelstudioTokenPlan,
5242 base_url,
5243 "deepseek-v4-pro",
5244 None,
5245 ));
5246 }
5247
5248 #[test]
5249 fn modelstudio_coding_plan_chat_route_is_classified_for_all_supported_identities() {
5250 // The picker represents Coding Plan as mode = "coding-plan" under
5251 // the primary provider id, so the chat client receives
5252 // ModelstudioTokenPlan with the Coding Plan URL. Direct configuration
5253 // also retains the legacy ModelstudioCodingPlan identity.
5254 let base_url = crate::config::DEFAULT_MODELSTUDIO_CODING_PLAN_BASE_URL;
5255 for provider in [
5256 ProviderKind::ModelstudioTokenPlan,
5257 ProviderKind::ModelstudioCodingPlan,
5258 ] {
5259 let mut body = json!({});
5260 apply_route_reasoning_controls(
5261 &mut body,
5262 provider,
5263 base_url,
5264 "qwen3.7-plus",
5265 Some("high"),
5266 );
5267
5268 assert_eq!(body["enable_thinking"], json!(true), "{provider:?}");
5269 assert_eq!(body["preserve_thinking"], json!(true), "{provider:?}");
5270 assert_eq!(
5271 reasoning_stream_style_for_route(provider, base_url, "qwen3.7-plus", None),
5272 ReasoningStreamStyle::SeparateField,
5273 "{provider:?}",
5274 );
5275 assert!(should_replay_reasoning_content_for_provider_on_route(
5276 provider,
5277 base_url,
5278 "qwen3.7-plus",
5279 None,
5280 ));
5281 }
5282 }
5283
5284 #[test]
5285 fn modelstudio_workspace_scoped_token_plan_route_is_recognized() {
5286 let workspace_url =
5287 "https://workspace-123.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
5288 // Stream classification works on the workspace-scoped host...
5289 assert_eq!(
5290 reasoning_stream_style_for_route(
5291 ProviderKind::ModelstudioTokenPlan,
5292 workspace_url,
5293 "qwen3.8-max",
5294 None,
5295 ),
5296 ReasoningStreamStyle::SeparateField,
5297 );
5298 assert_eq!(
5299 reasoning_stream_style_for_route(
5300 ProviderKind::ModelstudioTokenPlan,
5301 workspace_url,
5302 "qwen3.7-plus",
5303 None,
5304 ),
5305 ReasoningStreamStyle::SeparateField,
5306 );
5307 // ...and the route gate decides replay the same way it does on the
5308 // default host: qwen3.8 never replays, documented preserve models do.
5309 assert!(!should_replay_reasoning_content_for_provider_on_route(
5310 ProviderKind::ModelstudioTokenPlan,
5311 workspace_url,
5312 "qwen3.8-max",
5313 None,
5314 ));
5315 assert!(should_replay_reasoning_content_for_provider_on_route(
5316 ProviderKind::ModelstudioTokenPlan,
5317 workspace_url,
5318 "qwen3.7-plus",
5319 None,
5320 ));
5321 let mut body = json!({});
5322 apply_route_reasoning_controls(
5323 &mut body,
5324 ProviderKind::ModelstudioTokenPlan,
5325 workspace_url,
5326 "qwen3.7-plus",
5327 Some("high"),
5328 );
5329 assert_eq!(body["preserve_thinking"], json!(true), "{body}");
5330 }
5331
5332 #[test]
5333 fn modelstudio_thinking_named_unknown_model_cannot_bypass_the_route_gate() {
5334 // An unknown Model Studio model whose *name* suggests reasoning must
5335 // not gain replay: only the documented `preserve_thinking` list (plus
5336 // the concrete DeepSeek V4 family ids) authorizes historical
5337 // `reasoning_content` on exact Model Studio routes. The generic
5338 // `-thinking`/`reasoner` heuristics prove nothing about DashScope's
5339 // request dialect.
5340 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5341 for model in [
5342 "foo-thinking",
5343 "foo-reasoner",
5344 "new-model-reasoning",
5345 "acme-reasoner-v2",
5346 ] {
5347 assert!(
5348 !should_replay_reasoning_content_for_provider_on_route(
5349 ProviderKind::ModelstudioTokenPlan,
5350 base_url,
5351 model,
5352 None,
5353 ),
5354 "{model}"
5355 );
5356 // The shaper writes no reasoning controls for it either — fail
5357 // closed on the wire.
5358 let mut body = json!({ "enable_thinking": true, "preserve_thinking": true });
5359 apply_route_reasoning_controls(
5360 &mut body,
5361 ProviderKind::ModelstudioTokenPlan,
5362 base_url,
5363 model,
5364 Some("high"),
5365 );
5366 assert!(body.get("enable_thinking").is_none(), "{model}: {body}");
5367 assert!(body.get("preserve_thinking").is_none(), "{model}: {body}");
5368 }
5369 // A suggestive name is not a replay contract on any route.
5370 assert!(!requires_reasoning_content("foo-thinking"));
5371 assert!(!requires_reasoning_content("foo-reasoner"));
5372 }
5373
5374 #[test]
5375 fn modelstudio_qwen36_flash_preserves_thinking_per_current_documentation() {
5376 // Current Alibaba documentation explicitly includes qwen3.6-flash
5377 // (and its documented snapshots) among the `preserve_thinking`
5378 // models. Keep positive coverage so any future narrowing of the
5379 // preserve list has to remove it deliberately.
5380 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5381 for model in ["qwen3.6-flash", "qwen3.6-flash-2026-04-16"] {
5382 assert!(should_replay_reasoning_content_for_provider_on_route(
5383 ProviderKind::ModelstudioTokenPlan,
5384 base_url,
5385 model,
5386 None,
5387 ));
5388 let mut body = json!({});
5389 apply_route_reasoning_controls(
5390 &mut body,
5391 ProviderKind::ModelstudioTokenPlan,
5392 base_url,
5393 model,
5394 Some("high"),
5395 );
5396 assert_eq!(body["enable_thinking"], json!(true), "{model}");
5397 assert_eq!(body["preserve_thinking"], json!(true), "{model}: {body}");
5398 // Hybrid semantics: an explicit off disables both, and replay
5399 // follows the effort gate.
5400 assert!(!should_replay_reasoning_content_for_provider_on_route(
5401 ProviderKind::ModelstudioTokenPlan,
5402 base_url,
5403 model,
5404 Some("off"),
5405 ));
5406 let mut off_body = json!({});
5407 apply_route_reasoning_controls(
5408 &mut off_body,
5409 ProviderKind::ModelstudioTokenPlan,
5410 base_url,
5411 model,
5412 Some("off"),
5413 );
5414 assert_eq!(off_body["enable_thinking"], json!(false), "{model}");
5415 assert_eq!(off_body["preserve_thinking"], json!(false), "{model}");
5416 }
5417
5418 for invented in [
5419 "qwen3.6-flash-future",
5420 "qwen3.7-plus-proxy",
5421 "kimi-k2.7-code-unverified",
5422 ] {
5423 assert!(
5424 !should_replay_reasoning_content_for_provider_on_route(
5425 ProviderKind::ModelstudioTokenPlan,
5426 base_url,
5427 invented,
5428 None,
5429 ),
5430 "prefix lookalikes must fail closed: {invented}"
5431 );
5432 }
5433 }
5434
5435 #[test]
5436 fn modelstudio_kimi_k27_code_is_thinking_only_and_preserves_trace() {
5437 // NOTE: unlike the qwen3.8 pair, this classification is asserted by
5438 // PR #5233 rather than corroborated by models_dev.bundled.json, which
5439 // lists kimi-k2.7-code with `reasoning: true` and no `always_on`.
5440 let base_url = crate::config::DEFAULT_MODELSTUDIO_TOKEN_PLAN_BASE_URL;
5441 for model in [
5442 "kimi-k2.7-code",
5443 "kimi/kimi-k2.7-code",
5444 "kimi/kimi-k2.7-code-highspeed",
5445 ] {
5446 let mut body = json!({});
5447 apply_route_reasoning_controls(
5448 &mut body,
5449 ProviderKind::ModelstudioTokenPlan,
5450 base_url,
5451 model,
5452 Some("off"),
5453 );
5454
5455 assert!(body.get("enable_thinking").is_none(), "{model}: {body}");
5456 assert_eq!(body["preserve_thinking"], json!(true), "{model}");
5457 assert_eq!(
5458 reasoning_stream_style_for_route(
5459 ProviderKind::ModelstudioTokenPlan,
5460 base_url,
5461 model,
5462 None,
5463 ),
5464 ReasoningStreamStyle::SeparateField,
5465 "{model}",
5466 );
5467 assert!(should_replay_reasoning_content_for_provider_on_route(
5468 ProviderKind::ModelstudioTokenPlan,
5469 base_url,
5470 model,
5471 Some("off"),
5472 ));
5473 }
5474 }
5475
5476 #[test]
5477 fn stream_classifies_moonshot_kimi_as_reasoning() {
5478 // #3016: without this, Kimi thinking leaked into answer text.
5479 assert!(is_reasoning_model_for_stream(
5480 ProviderKind::Moonshot,
5481 "kimi-k2.6"
5482 ));
5483 assert!(
5484 is_reasoning_model_for_stream(ProviderKind::Moonshot, "kimi-for-coding"),
5485 "Kimi Code's stable model id now maps to K2.7 Code and streams reasoning_content"
5486 );
5487 }
5488
5489 #[test]
5490 fn moonshot_and_minimax_replay_reasoning_content_for_supported_models() {
5491 assert!(should_replay_reasoning_content_for_provider(
5492 ProviderKind::Moonshot,
5493 "kimi-k2.7-code",
5494 None,
5495 ));
5496 assert!(should_replay_reasoning_content_for_provider(
5497 ProviderKind::Moonshot,
5498 "kimi-for-coding",
5499 None,
5500 ));
5501 assert!(should_replay_reasoning_content_for_provider(
5502 ProviderKind::Minimax,
5503 "MiniMax-M3",
5504 None,
5505 ));
5506 assert!(should_replay_reasoning_content_for_provider(
5507 ProviderKind::Zai,
5508 "GLM-5.2",
5509 None,
5510 ));
5511 assert!(should_replay_reasoning_content_for_provider(
5512 ProviderKind::Zai,
5513 "GLM-5.3",
5514 None,
5515 ));
5516 assert!(!should_replay_reasoning_content_for_provider(
5517 ProviderKind::Moonshot,
5518 "kimi-for-coding",
5519 Some("off"),
5520 ));
5521 }
5522
5523 #[test]
5524 fn bare_k3_reasoning_semantics_are_scoped_to_exact_kimi_code_route() {
5525 let kimi_code = crate::config::DEFAULT_KIMI_CODE_BASE_URL;
5526 let direct_moonshot = crate::config::DEFAULT_MOONSHOT_BASE_URL;
5527
5528 assert!(should_replay_reasoning_content_for_provider_on_route(
5529 ProviderKind::Moonshot,
5530 kimi_code,
5531 crate::config::KIMI_CODE_K3_MODEL,
5532 Some("high"),
5533 ));
5534 assert_eq!(
5535 reasoning_stream_style_for_route(
5536 ProviderKind::Moonshot,
5537 kimi_code,
5538 crate::config::KIMI_CODE_K3_MODEL,
5539 None,
5540 ),
5541 ReasoningStreamStyle::SeparateField
5542 );
5543
5544 assert!(!should_replay_reasoning_content_for_provider_on_route(
5545 ProviderKind::Moonshot,
5546 direct_moonshot,
5547 crate::config::KIMI_CODE_K3_MODEL,
5548 Some("high"),
5549 ));
5550 // Display is not replay: a reasoning field that arrives on the
5551 // neighboring route still renders as Thinking (#6501), while the
5552 // replay contract above stays scoped to the exact membership route.
5553 assert_eq!(
5554 reasoning_stream_style_for_route(
5555 ProviderKind::Moonshot,
5556 direct_moonshot,
5557 crate::config::KIMI_CODE_K3_MODEL,
5558 None,
5559 ),
5560 ReasoningStreamStyle::SeparateField
5561 );
5562 assert!(
5563 should_replay_reasoning_content_for_provider_on_route(
5564 ProviderKind::Moonshot,
5565 kimi_code,
5566 crate::config::KIMI_CODE_K3_MODEL,
5567 Some("off"),
5568 ),
5569 "exact membership K3 stays always-thinking even for a stale raw Off caller"
5570 );
5571 }
5572
5573 #[test]
5574 fn direct_moonshot_k3_is_always_thinking_and_replays_reasoning() {
5575 let direct = crate::config::DEFAULT_MOONSHOT_BASE_URL;
5576 let model = crate::config::MOONSHOT_KIMI_K3_MODEL;
5577
5578 for effort in [Some("off"), Some("low"), Some("high"), Some("max"), None] {
5579 assert!(should_replay_reasoning_content_for_provider_on_route(
5580 ProviderKind::Moonshot,
5581 direct,
5582 model,
5583 effort,
5584 ));
5585 }
5586 assert_eq!(
5587 reasoning_stream_style_for_route(ProviderKind::Moonshot, direct, model, None),
5588 ReasoningStreamStyle::SeparateField
5589 );
5590 }
5591
5592 #[test]
5593 fn xiaomi_mimo_uses_max_completion_tokens_payload_key() {
5594 let mut body = json!({
5595 "model": "mimo-v2.5-pro",
5596 "messages": [],
5597 "max_tokens": 8192,
5598 });
5599
5600 apply_provider_token_limit(
5601 &mut body,
5602 ProviderKind::XiaomiMimo,
5603 "https://api.xiaomimimo.com/v1",
5604 "mimo-v2.5-pro",
5605 8192,
5606 );
5607
5608 assert!(body.get("max_tokens").is_none());
5609 assert_eq!(
5610 body.get("max_completion_tokens")
5611 .and_then(serde_json::Value::as_u64),
5612 Some(8192)
5613 );
5614 }
5615
5616 #[test]
5617 fn openai_reasoning_model_uses_completion_token_limit_and_effort_field() {
5618 let mut body = json!({
5619 "model": "gpt-5.5",
5620 "messages": [],
5621 "max_tokens": 4096,
5622 });
5623
5624 apply_provider_token_limit(
5625 &mut body,
5626 ProviderKind::Openai,
5627 "https://api.openai.com/v1",
5628 "gpt-5.5",
5629 4096,
5630 );
5631 apply_openai_reasoning_effort(&mut body, ProviderKind::Openai, "gpt-5.5", Some("high"));
5632
5633 assert!(body.get("max_tokens").is_none());
5634 assert_eq!(
5635 body.get("max_completion_tokens")
5636 .and_then(serde_json::Value::as_u64),
5637 Some(4096)
5638 );
5639 assert_eq!(
5640 body.get("reasoning_effort")
5641 .and_then(serde_json::Value::as_str),
5642 Some("high")
5643 );
5644 }
5645
5646 #[test]
5647 fn gpt_56_uses_documented_max_reasoning_effort() {
5648 let mut body = json!({
5649 "model": "gpt-5.6-sol",
5650 "messages": [],
5651 "max_tokens": 8192,
5652 });
5653
5654 apply_provider_token_limit(
5655 &mut body,
5656 ProviderKind::Openai,
5657 "https://api.openai.com/v1",
5658 "gpt-5.6-sol",
5659 8192,
5660 );
5661 apply_openai_reasoning_effort(&mut body, ProviderKind::Openai, "gpt-5.6-sol", Some("max"));
5662
5663 assert!(body.get("max_tokens").is_none());
5664 assert_eq!(body["max_completion_tokens"], json!(8192));
5665 assert_eq!(body["reasoning_effort"], json!("max"));
5666 }
5667
5668 #[test]
5669 fn grok_47_and_46_use_exact_first_party_reasoning_effort_ladder() {
5670 // Both catalog rows document low/medium/high/xhigh; reasoning cannot
5671 // be disabled, so `off` is the documented default `high`.
5672 for model in [
5673 crate::config::XAI_GROK_4_7_MODEL,
5674 crate::config::XAI_GROK_4_6_MODEL,
5675 ] {
5676 for (requested, expected) in [
5677 ("off", "high"),
5678 ("low", "low"),
5679 ("medium", "medium"),
5680 ("high", "high"),
5681 ("xhigh", "xhigh"),
5682 ("max", "xhigh"),
5683 ] {
5684 let mut body = json!({});
5685 apply_route_reasoning_controls(
5686 &mut body,
5687 ProviderKind::Xai,
5688 crate::config::DEFAULT_XAI_BASE_URL,
5689 model,
5690 Some(requested),
5691 );
5692 assert_eq!(
5693 body,
5694 json!({ "reasoning_effort": expected }),
5695 "{model} {requested}"
5696 );
5697 }
5698 }
5699
5700 // A row without a documented effort option (grok-4.3) and an id the
5701 // catalog does not know send nothing.
5702 for model in [crate::config::XAI_GROK_4_3_MODEL, "grok-9-unknown"] {
5703 let mut body = json!({});
5704 apply_route_reasoning_controls(
5705 &mut body,
5706 ProviderKind::Xai,
5707 crate::config::DEFAULT_XAI_BASE_URL,
5708 model,
5709 Some("medium"),
5710 );
5711 assert_eq!(body, json!({}), "{model}");
5712 }
5713
5714 let mut provider_default = json!({});
5715 apply_route_reasoning_controls(
5716 &mut provider_default,
5717 ProviderKind::Xai,
5718 crate::config::DEFAULT_XAI_BASE_URL,
5719 crate::config::XAI_GROK_4_6_MODEL,
5720 Some("auto"),
5721 );
5722 assert_eq!(provider_default, json!({}));
5723
5724 let mut custom = json!({});
5725 apply_route_reasoning_controls(
5726 &mut custom,
5727 ProviderKind::Xai,
5728 "https://gateway.example/v1",
5729 crate::config::XAI_GROK_4_6_MODEL,
5730 Some("medium"),
5731 );
5732 assert_eq!(custom, json!({}));
5733 }
5734
5735 #[test]
5736 fn grok_45_uses_first_party_ladder_and_maps_xhigh_to_high() {
5737 for (requested, expected) in [
5738 ("off", "high"),
5739 ("low", "low"),
5740 ("medium", "medium"),
5741 ("high", "high"),
5742 ("xhigh", "high"),
5743 ("max", "high"),
5744 ] {
5745 let mut body = json!({});
5746 apply_route_reasoning_controls(
5747 &mut body,
5748 ProviderKind::Xai,
5749 crate::config::DEFAULT_XAI_BASE_URL,
5750 crate::config::XAI_GROK_4_5_MODEL,
5751 Some(requested),
5752 );
5753 assert_eq!(body, json!({ "reasoning_effort": expected }), "{requested}");
5754 }
5755 }
5756
5757 #[test]
5758 fn inkling_uses_its_exact_reasoning_vocabulary_without_thinking_extension() {
5759 for (requested, expected) in [
5760 ("off", "none"),
5761 ("minimal", "minimal"),
5762 ("low", "low"),
5763 ("medium", "medium"),
5764 ("high", "high"),
5765 ("max", "max"),
5766 ("xhigh", "max"),
5767 ] {
5768 let mut body = json!({
5769 "thinking": { "type": "enabled" },
5770 "reasoning_effort": "xhigh",
5771 });
5772
5773 apply_inkling_reasoning_effort(
5774 &mut body,
5775 ProviderKind::Together,
5776 "thinkingmachines/inkling",
5777 Some(requested),
5778 );
5779
5780 assert_eq!(body["reasoning_effort"], json!(expected));
5781 assert!(body.get("thinking").is_none());
5782 }
5783 }
5784
5785 #[test]
5786 fn inkling_reasoning_override_is_scoped_to_the_exact_together_route() {
5787 let mut other_model = json!({ "thinking": { "type": "enabled" } });
5788 apply_inkling_reasoning_effort(
5789 &mut other_model,
5790 ProviderKind::Together,
5791 "deepseek-ai/DeepSeek-V4-Pro",
5792 Some("max"),
5793 );
5794 assert_eq!(other_model["thinking"]["type"], json!("enabled"));
5795 assert!(other_model.get("reasoning_effort").is_none());
5796
5797 let mut other_provider = json!({ "thinking": { "type": "enabled" } });
5798 apply_inkling_reasoning_effort(
5799 &mut other_provider,
5800 ProviderKind::Openrouter,
5801 "thinkingmachines/inkling",
5802 Some("max"),
5803 );
5804 assert_eq!(other_provider["thinking"]["type"], json!("enabled"));
5805 assert!(other_provider.get("reasoning_effort").is_none());
5806 }
5807
5808 #[test]
5809 fn kimi_code_k3_uses_documented_nested_thinking_effort() {
5810 for (requested, expected) in [
5811 ("low", json!({ "type": "enabled", "effort": "low" })),
5812 ("minimum", json!({ "type": "enabled", "effort": "low" })),
5813 ("light", json!({ "type": "enabled", "effort": "low" })),
5814 ("medium", json!({ "type": "enabled", "effort": "high" })),
5815 ("high", json!({ "type": "enabled", "effort": "high" })),
5816 ("xhigh", json!({ "type": "enabled", "effort": "max" })),
5817 ("ultra", json!({ "type": "enabled", "effort": "max" })),
5818 ("max", json!({ "type": "enabled", "effort": "max" })),
5819 ("none", json!({ "type": "enabled", "effort": "low" })),
5820 ("off", json!({ "type": "enabled", "effort": "low" })),
5821 ] {
5822 let mut body = json!({ "reasoning_effort": "stale" });
5823 apply_kimi_code_k3_reasoning_effort(
5824 &mut body,
5825 ProviderKind::Moonshot,
5826 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
5827 crate::config::KIMI_CODE_K3_MODEL,
5828 Some(requested),
5829 );
5830
5831 assert_eq!(body["thinking"], expected, "requested {requested}");
5832 assert!(body.get("reasoning_effort").is_none());
5833 }
5834 }
5835
5836 #[test]
5837 fn kimi_code_k3_256k_uses_k3_reasoning_and_membership_sampling_contracts() {
5838 let mut body = json!({
5839 "reasoning_effort": "stale",
5840 "temperature": 0.3,
5841 "top_p": 0.8,
5842 });
5843 apply_kimi_code_k3_reasoning_effort(
5844 &mut body,
5845 ProviderKind::Moonshot,
5846 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
5847 crate::config::KIMI_CODE_K3_256K_MODEL,
5848 Some("max"),
5849 );
5850 apply_kimi_code_fixed_sampling(
5851 &mut body,
5852 ProviderKind::Moonshot,
5853 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
5854 crate::config::KIMI_CODE_K3_256K_MODEL,
5855 );
5856
5857 assert_eq!(
5858 body["thinking"],
5859 json!({ "type": "enabled", "effort": "max" })
5860 );
5861 assert!(body.get("reasoning_effort").is_none());
5862 assert!(body.get("temperature").is_none());
5863 assert!(body.get("top_p").is_none());
5864 }
5865
5866 #[test]
5867 fn kimi_code_fixed_sampling_does_not_leak_to_neighbor_routes() {
5868 for (provider, base_url, model) in [
5869 (
5870 ProviderKind::Moonshot,
5871 crate::config::DEFAULT_MOONSHOT_BASE_URL,
5872 crate::config::KIMI_CODE_K3_256K_MODEL,
5873 ),
5874 (
5875 ProviderKind::Openrouter,
5876 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
5877 crate::config::KIMI_CODE_K3_256K_MODEL,
5878 ),
5879 (
5880 ProviderKind::Moonshot,
5881 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
5882 "k3-256k-preview",
5883 ),
5884 ] {
5885 let mut body = json!({ "temperature": 0.3, "top_p": 0.8 });
5886 apply_kimi_code_fixed_sampling(&mut body, provider, base_url, model);
5887 assert_eq!(body["temperature"], json!(0.3));
5888 assert_eq!(body["top_p"], json!(0.8));
5889 }
5890 }
5891
5892 #[test]
5893 fn direct_moonshot_k3_uses_top_level_effort_and_never_disables_thinking() {
5894 for (requested, expected) in [
5895 ("off", "low"),
5896 ("none", "low"),
5897 ("low", "low"),
5898 ("medium", "high"),
5899 ("high", "high"),
5900 ("xhigh", "max"),
5901 ("max", "max"),
5902 ] {
5903 let mut body = json!({
5904 "model": crate::config::MOONSHOT_KIMI_K3_MODEL,
5905 "thinking": { "type": "disabled" },
5906 });
5907 apply_route_reasoning_controls(
5908 &mut body,
5909 ProviderKind::Moonshot,
5910 crate::config::DEFAULT_MOONSHOT_BASE_URL,
5911 crate::config::MOONSHOT_KIMI_K3_MODEL,
5912 Some(requested),
5913 );
5914
5915 assert_eq!(body["reasoning_effort"], json!(expected), "{requested}");
5916 assert!(body.get("thinking").is_none(), "{requested}: {body}");
5917 }
5918
5919 let mut provider_default = json!({ "thinking": { "type": "enabled" } });
5920 apply_route_reasoning_controls(
5921 &mut provider_default,
5922 ProviderKind::Moonshot,
5923 crate::config::DEFAULT_MOONSHOT_BASE_URL,
5924 crate::config::MOONSHOT_KIMI_K3_MODEL,
5925 Some("auto"),
5926 );
5927 assert!(provider_default.get("thinking").is_none());
5928 assert!(provider_default.get("reasoning_effort").is_none());
5929 }
5930
5931 #[test]
5932 fn direct_moonshot_k3_uses_modern_token_field_and_fixed_sampling_only_on_exact_route() {
5933 let mut direct = json!({
5934 "max_tokens": 64,
5935 "temperature": 0.2,
5936 "top_p": 0.9,
5937 });
5938 apply_provider_token_limit(
5939 &mut direct,
5940 ProviderKind::Moonshot,
5941 crate::config::DEFAULT_MOONSHOT_BASE_URL,
5942 crate::config::MOONSHOT_KIMI_K3_MODEL,
5943 64,
5944 );
5945 apply_direct_moonshot_k3_fixed_sampling(
5946 &mut direct,
5947 ProviderKind::Moonshot,
5948 crate::config::DEFAULT_MOONSHOT_BASE_URL,
5949 crate::config::MOONSHOT_KIMI_K3_MODEL,
5950 );
5951 assert_eq!(direct["max_completion_tokens"], json!(64));
5952 assert!(direct.get("max_tokens").is_none());
5953 assert!(direct.get("temperature").is_none());
5954 assert!(direct.get("top_p").is_none());
5955
5956 let mut neighbor = json!({
5957 "max_tokens": 64,
5958 "temperature": 0.2,
5959 "top_p": 0.9,
5960 });
5961 apply_provider_token_limit(
5962 &mut neighbor,
5963 ProviderKind::Moonshot,
5964 "https://proxy.example/v1",
5965 crate::config::MOONSHOT_KIMI_K3_MODEL,
5966 64,
5967 );
5968 apply_direct_moonshot_k3_fixed_sampling(
5969 &mut neighbor,
5970 ProviderKind::Moonshot,
5971 "https://proxy.example/v1",
5972 crate::config::MOONSHOT_KIMI_K3_MODEL,
5973 );
5974 assert_eq!(neighbor["max_tokens"], json!(64));
5975 assert!(neighbor.get("max_completion_tokens").is_none());
5976 assert_eq!(neighbor["temperature"], json!(0.2));
5977 assert_eq!(neighbor["top_p"], json!(0.9));
5978 }
5979
5980 #[test]
5981 fn direct_and_membership_k3_reasoning_dialects_do_not_cross_routes() {
5982 let mut membership = json!({});
5983 apply_route_reasoning_controls(
5984 &mut membership,
5985 ProviderKind::Moonshot,
5986 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
5987 crate::config::KIMI_CODE_K3_MODEL,
5988 Some("max"),
5989 );
5990 assert_eq!(
5991 membership["thinking"],
5992 json!({ "type": "enabled", "effort": "max" })
5993 );
5994 assert!(membership.get("reasoning_effort").is_none());
5995
5996 for (base_url, model) in [
5997 (
5998 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
5999 crate::config::MOONSHOT_KIMI_K3_MODEL,
6000 ),
6001 (
6002 crate::config::DEFAULT_MOONSHOT_BASE_URL,
6003 crate::config::KIMI_CODE_K3_MODEL,
6004 ),
6005 (
6006 "https://proxy.example/v1",
6007 crate::config::MOONSHOT_KIMI_K3_MODEL,
6008 ),
6009 ] {
6010 let mut neighbor = json!({});
6011 apply_route_reasoning_controls(
6012 &mut neighbor,
6013 ProviderKind::Moonshot,
6014 base_url,
6015 model,
6016 Some("max"),
6017 );
6018 assert_eq!(
6019 neighbor["thinking"],
6020 json!({ "type": "enabled" }),
6021 "{base_url} / {model}"
6022 );
6023 assert!(neighbor.get("reasoning_effort").is_none());
6024 assert!(neighbor.pointer("/thinking/effort").is_none());
6025 }
6026 }
6027
6028 #[test]
6029 fn kimi_code_k3_effort_override_never_leaks_to_neighbor_routes() {
6030 for (base_url, model) in [
6031 (crate::config::DEFAULT_KIMI_CODE_BASE_URL, "kimi-k3"),
6032 (
6033 crate::config::DEFAULT_KIMI_CODE_BASE_URL,
6034 crate::config::DEFAULT_KIMI_CODE_MODEL,
6035 ),
6036 (crate::config::DEFAULT_MOONSHOT_BASE_URL, "k3"),
6037 ] {
6038 let mut body = json!({ "thinking": { "type": "enabled" } });
6039 apply_kimi_code_k3_reasoning_effort(
6040 &mut body,
6041 ProviderKind::Moonshot,
6042 base_url,
6043 model,
6044 Some("max"),
6045 );
6046
6047 assert_eq!(body["thinking"], json!({ "type": "enabled" }));
6048 assert!(
6049 body.pointer("/thinking/effort").is_none(),
6050 "{base_url} / {model}"
6051 );
6052 assert!(body.get("reasoning_effort").is_none());
6053 }
6054 }
6055
6056 #[test]
6057 fn muse_spark_uses_meta_reasoning_effort_without_openai_token_rewrite() {
6058 let mut body = json!({
6059 "model": "muse-spark-1.1",
6060 "messages": [],
6061 "max_tokens": 8192,
6062 });
6063
6064 apply_provider_token_limit(
6065 &mut body,
6066 ProviderKind::Meta,
6067 "https://api.meta.ai/v1",
6068 "muse-spark-1.1",
6069 8192,
6070 );
6071 apply_openai_reasoning_effort(&mut body, ProviderKind::Meta, "muse-spark-1.1", Some("max"));
6072
6073 assert_eq!(body["max_tokens"], json!(8192));
6074 assert!(body.get("max_completion_tokens").is_none());
6075 assert_eq!(body["reasoning_effort"], json!("xhigh"));
6076 }
6077
6078 #[test]
6079 fn provider_regression_5853_muse_family_preserves_reasoning_effort() {
6080 for model in [
6081 "muse-spark-1.2",
6082 "muse-spark-1.3",
6083 "muse-spark-1.3-contributor",
6084 " MUSE-SPARK-1.3 ",
6085 ] {
6086 for (effort, wire) in [
6087 ("low", "low"),
6088 ("medium", "medium"),
6089 ("high", "high"),
6090 ("xhigh", "xhigh"),
6091 ("max", "xhigh"),
6092 ("ultra", "xhigh"),
6093 ] {
6094 let mut body = json!({"model": model});
6095 apply_openai_reasoning_effort(&mut body, ProviderKind::Meta, model, Some(effort));
6096 assert_eq!(body["reasoning_effort"], wire, "{model}, effort={effort}");
6097 }
6098 }
6099 for (provider, model, effort) in [
6100 (ProviderKind::Openai, "muse-spark-1.3", Some("high")),
6101 (ProviderKind::Meta, "muse-glimmer-fixture", Some("high")),
6102 (ProviderKind::Meta, "muse-sparks-fixture", Some("high")),
6103 (ProviderKind::Meta, "muse-spark-1.3", None),
6104 ] {
6105 let mut body = json!({"model": model});
6106 apply_openai_reasoning_effort(&mut body, provider, model, effort);
6107 assert!(
6108 body.get("reasoning_effort").is_none(),
6109 "{provider:?}, {model}"
6110 );
6111 }
6112 }
6113
6114 #[test]
6115 fn openai_non_reasoning_model_omits_reasoning_only_fields() {
6116 let mut body = json!({
6117 "model": "gpt-4o",
6118 "messages": [],
6119 "max_tokens": 4096,
6120 });
6121
6122 apply_provider_token_limit(
6123 &mut body,
6124 ProviderKind::Openai,
6125 "https://api.openai.com/v1",
6126 "gpt-4o",
6127 4096,
6128 );
6129 apply_openai_reasoning_effort(&mut body, ProviderKind::Openai, "gpt-4o", Some("high"));
6130
6131 assert_eq!(
6132 body.get("max_tokens").and_then(serde_json::Value::as_u64),
6133 Some(4096)
6134 );
6135 assert!(body.get("max_completion_tokens").is_none());
6136 assert!(body.get("reasoning_effort").is_none());
6137 }
6138
6139 #[test]
6140 fn openai_provider_deepseek_compatible_model_keeps_chat_token_field() {
6141 let mut body = json!({
6142 "model": "deepseek-v4-pro",
6143 "messages": [],
6144 "max_tokens": 4096,
6145 });
6146
6147 apply_provider_token_limit(
6148 &mut body,
6149 ProviderKind::Openai,
6150 "https://api.openai.com/v1",
6151 "deepseek-v4-pro",
6152 4096,
6153 );
6154 apply_openai_reasoning_effort(
6155 &mut body,
6156 ProviderKind::Openai,
6157 "deepseek-v4-pro",
6158 Some("high"),
6159 );
6160
6161 assert_eq!(
6162 body.get("max_tokens").and_then(serde_json::Value::as_u64),
6163 Some(4096)
6164 );
6165 assert!(body.get("max_completion_tokens").is_none());
6166 assert!(body.get("reasoning_effort").is_none());
6167 }
6168
6169 #[test]
6170 fn deepseek_model_on_openai_provider_still_replays_reasoning_content() {
6171 // #1739 / #1694: a DeepSeek thinking model pointed at a
6172 // DeepSeek-compatible endpoint via the generic `openai` provider must
6173 // still replay reasoning_content, even though the provider itself does
6174 // not accept the field. Otherwise the thinking-mode API returns 400.
6175 assert!(should_replay_reasoning_content_for_provider(
6176 ProviderKind::Openai,
6177 "deepseek-v4-flash",
6178 None,
6179 ));
6180 assert!(should_replay_reasoning_content_for_provider(
6181 ProviderKind::Openai,
6182 "deepseek-v4-pro",
6183 None,
6184 ));
6185 assert!(should_replay_reasoning_content_for_provider(
6186 ProviderKind::Openai,
6187 "deepseek-reasoner",
6188 Some("medium"),
6189 ));
6190 // The documented escape hatch still wins over model detection.
6191 assert!(!should_replay_reasoning_content_for_provider(
6192 ProviderKind::Openai,
6193 "deepseek-v4-flash",
6194 Some("off"),
6195 ));
6196 }
6197
6198 #[test]
6199 fn generic_model_on_openai_provider_still_strips_reasoning_content() {
6200 // #1542 no-regression guard: a genuine non-DeepSeek model on the
6201 // openai provider must continue to have reasoning_content stripped.
6202 assert!(!should_replay_reasoning_content_for_provider(
6203 ProviderKind::Openai,
6204 "qwen3-coder",
6205 None,
6206 ));
6207 assert!(!should_replay_reasoning_content_for_provider(
6208 ProviderKind::Openai,
6209 "claude-sonnet-4-6",
6210 None,
6211 ));
6212 }
6213
6214 #[test]
6215 fn suggestive_unknown_model_names_never_authorize_reasoning_replay() {
6216 for provider in [
6217 ProviderKind::Openai,
6218 ProviderKind::Deepseek,
6219 ProviderKind::Openrouter,
6220 ProviderKind::Moonshot,
6221 ProviderKind::Zai,
6222 ] {
6223 for model in [
6224 "foo-thinking",
6225 "foo-reasoner",
6226 "acme-reasoning",
6227 "future-reasoner-v9",
6228 ] {
6229 assert!(
6230 !should_replay_reasoning_content_for_provider(provider, model, None),
6231 "{provider:?} {model}"
6232 );
6233 }
6234 }
6235 }
6236
6237 #[test]
6238 fn stream_classifies_deepseek_model_on_openai_provider_as_reasoning() {
6239 // #1739: the SSE parser must treat a DeepSeek thinking model on the
6240 // generic `openai` provider (DeepSeek-compatible endpoint) as a
6241 // reasoning model, or incoming `reasoning_content` tokens are stored
6242 // as answer text and the subsequent replay still 400s.
6243 assert!(is_reasoning_model_for_stream(
6244 ProviderKind::Openai,
6245 "deepseek-v4-flash"
6246 ));
6247 assert!(is_reasoning_model_for_stream(
6248 ProviderKind::Openai,
6249 "deepseek-v4-pro"
6250 ));
6251 assert!(is_reasoning_model_for_stream(
6252 ProviderKind::Openai,
6253 "deepseek-reasoner"
6254 ));
6255 // Native DeepSeek provider was already correct; stays correct.
6256 assert!(is_reasoning_model_for_stream(
6257 ProviderKind::Deepseek,
6258 "deepseek-v4-pro"
6259 ));
6260 }
6261
6262 #[test]
6263 fn zai_tiered_effort_applies_to_glm_5_2_and_glm_5_3_but_not_5_1() {
6264 let zai = crate::config::DEFAULT_ZAI_BASE_URL;
6265 // GLM-5.3 and GLM-5.3-Flash inherit GLM-5.2's reasoning_options
6266 // (effort high/max), so they must take the same tiered wire path.
6267 for model in [
6268 crate::config::ZAI_GLM_5_2_MODEL,
6269 crate::config::ZAI_GLM_5_3_MODEL,
6270 crate::config::ZAI_GLM_5_3_FLASH_MODEL,
6271 ] {
6272 let mut body = json!({});
6273 apply_route_reasoning_controls(&mut body, ProviderKind::Zai, zai, model, Some("max"));
6274 assert_eq!(body["reasoning_effort"], json!("max"), "{model} at max");
6275
6276 let mut body = json!({});
6277 apply_route_reasoning_controls(&mut body, ProviderKind::Zai, zai, model, Some("high"));
6278 assert_eq!(body["reasoning_effort"], json!("high"), "{model} at high");
6279 }
6280
6281 // GLM-5.1 and GLM-5-Turbo keep only the generic thinking control.
6282 for model in [
6283 crate::config::ZAI_GLM_5_1_MODEL,
6284 crate::config::ZAI_GLM_5_TURBO_MODEL,
6285 ] {
6286 let mut body = json!({});
6287 apply_route_reasoning_controls(&mut body, ProviderKind::Zai, zai, model, Some("max"));
6288 assert!(
6289 body.get("reasoning_effort").is_none(),
6290 "{model} must not receive tiered effort"
6291 );
6292 }
6293
6294 // A compatible gateway is not evidence of the Z.ai dialect, for 5.3
6295 // exactly as for 5.2.
6296 let mut body = json!({"thinking": {"type": "enabled"}});
6297 apply_route_reasoning_controls(
6298 &mut body,
6299 ProviderKind::Zai,
6300 "https://gateway.example.com/v1",
6301 crate::config::ZAI_GLM_5_3_MODEL,
6302 Some("max"),
6303 );
6304 assert!(body.get("reasoning_effort").is_none());
6305 assert!(body.get("thinking").is_none());
6306 }
6307
6308 #[test]
6309 fn zai_forced_thinking_models_never_send_thinking_disabled() {
6310 // BigModel and Z.ai document GLM-5.3 / GLM-5.3-Flash as forced-thinking:
6311 // `thinking.type: "disabled"` errors, effort accepts only low/high/max,
6312 // and the migration note for a former `disabled` payload is
6313 // `enabled` + `reasoning_effort: "low"`. Both hosts of the first-party
6314 // open platform — api.z.ai and open.bigmodel.cn — get the rewrite.
6315 for zai in [
6316 crate::config::DEFAULT_ZAI_BASE_URL,
6317 "https://open.bigmodel.cn/api/paas/v4",
6318 ] {
6319 for model in [
6320 crate::config::ZAI_GLM_5_3_MODEL,
6321 crate::config::ZAI_GLM_5_3_FLASH_MODEL,
6322 ] {
6323 let mut body = json!({});
6324 apply_route_reasoning_controls(
6325 &mut body,
6326 ProviderKind::Zai,
6327 zai,
6328 model,
6329 Some("off"),
6330 );
6331 assert_eq!(
6332 body["thinking"]["type"],
6333 json!("enabled"),
6334 "{model} must not send the rejected disabled toggle"
6335 );
6336 assert_eq!(
6337 body["reasoning_effort"],
6338 json!("low"),
6339 "{model} off becomes low"
6340 );
6341
6342 let mut body = json!({});
6343 apply_route_reasoning_controls(
6344 &mut body,
6345 ProviderKind::Zai,
6346 zai,
6347 model,
6348 Some("low"),
6349 );
6350 assert_eq!(
6351 body["reasoning_effort"],
6352 json!("low"),
6353 "{model} low is native"
6354 );
6355
6356 let mut body = json!({});
6357 apply_route_reasoning_controls(
6358 &mut body,
6359 ProviderKind::Zai,
6360 zai,
6361 model,
6362 Some("medium"),
6363 );
6364 assert_eq!(
6365 body["reasoning_effort"],
6366 json!("high"),
6367 "{model} medium maps to high"
6368 );
6369
6370 let mut body = json!({});
6371 apply_route_reasoning_controls(
6372 &mut body,
6373 ProviderKind::Zai,
6374 zai,
6375 model,
6376 Some("max"),
6377 );
6378 assert_eq!(
6379 body["reasoning_effort"],
6380 json!("max"),
6381 "{model} max stays max"
6382 );
6383
6384 // Unknown legacy values leave the field omitted so the API keeps
6385 // its documented default; nothing may reintroduce `disabled`.
6386 let mut body = json!({});
6387 apply_route_reasoning_controls(
6388 &mut body,
6389 ProviderKind::Zai,
6390 zai,
6391 model,
6392 Some("auto"),
6393 );
6394 assert!(
6395 body.get("reasoning_effort").is_none(),
6396 "{model} auto stays omitted"
6397 );
6398 assert_ne!(body["thinking"]["type"], json!("disabled"));
6399 }
6400
6401 // GLM-5.2 honours the generic disabled toggle on both hosts. On
6402 // BigModel that toggle now reaches the API (its docs still list
6403 // `disabled` for GLM-5.2) instead of being stripped by the
6404 // fail-closed gateway path.
6405 let mut body = json!({});
6406 apply_route_reasoning_controls(
6407 &mut body,
6408 ProviderKind::Zai,
6409 zai,
6410 crate::config::ZAI_GLM_5_2_MODEL,
6411 Some("off"),
6412 );
6413 assert_eq!(body["thinking"]["type"], json!("disabled"));
6414 assert!(body.get("reasoning_effort").is_none());
6415 }
6416 }
6417
6418 #[test]
6419 fn zai_bigmodel_adjacent_routes_stay_fail_closed() {
6420 // BigModel's `/preview` product and a plain-http neighbor are not the
6421 // documented Chat dialect, so they keep the gateway treatment: no
6422 // Z.ai reasoning fields, including on a forced-thinking model.
6423 for neighboring_route in [
6424 "https://open.bigmodel.cn/api/paas/v4/preview",
6425 "http://open.bigmodel.cn/api/paas/v4",
6426 ] {
6427 let mut body = json!({"thinking": {"type": "enabled"}});
6428 apply_route_reasoning_controls(
6429 &mut body,
6430 ProviderKind::Zai,
6431 neighboring_route,
6432 crate::config::ZAI_GLM_5_3_MODEL,
6433 Some("max"),
6434 );
6435 assert!(
6436 body.get("reasoning_effort").is_none(),
6437 "{neighboring_route} must not gain tiered effort"
6438 );
6439 assert!(
6440 body.get("thinking").is_none(),
6441 "{neighboring_route} must not keep the Z.ai thinking object"
6442 );
6443 }
6444 }
6445
6446 #[test]
6447 fn stream_classifies_known_large_reasoning_models_as_reasoning() {
6448 // Xiaomi MiMo and OpenRouter/Qwen/Trinity can stream private reasoning through a
6449 // `reasoning` delta without using a DeepSeek-looking model name. The
6450 // renderer must still route that field into Thinking cells instead
6451 // of plain assistant prose.
6452 assert!(
6453 is_reasoning_model_for_stream(ProviderKind::XiaomiMimo, "mimo-v2.5-pro"),
6454 "mimo-v2.5-pro should stream reasoning as thinking on Xiaomi MiMo"
6455 );
6456 assert!(
6457 is_reasoning_model_for_stream(ProviderKind::Arcee, "trinity-large-thinking"),
6458 "trinity-large-thinking should stream reasoning as thinking on direct Arcee"
6459 );
6460 assert!(
6461 is_reasoning_model_for_stream(ProviderKind::Zai, "GLM-5.2"),
6462 "GLM-5.2 should stream reasoning_content as thinking on direct Z.ai"
6463 );
6464 assert!(
6465 is_reasoning_model_for_stream(ProviderKind::Zai, "GLM-5.3"),
6466 "GLM-5.3 inherits GLM-5.2's reasoning capability on direct Z.ai"
6467 );
6468 for model in [
6469 "arcee-ai/trinity-large-thinking",
6470 "minimax/minimax-m3",
6471 "xiaomi/mimo-v2.5-pro",
6472 ] {
6473 assert!(
6474 is_reasoning_model_for_stream(ProviderKind::Openrouter, model),
6475 "{model} should stream reasoning as thinking on OpenRouter"
6476 );
6477 }
6478 }
6479
6480 #[test]
6481 fn stream_surfaces_reasoning_fields_on_every_unconfigured_route() {
6482 // #6501: grok-4.7 on xAI and a MiMo id newer than the offline catalog
6483 // streamed their reasoning as answer prose because the stream gate
6484 // required a provider allowlist AND a catalog row. The reasoning
6485 // fields are reasoning on every route that sends them; only an
6486 // explicit `reasoning_stream_style = "none"` restores pass-through.
6487 for (provider, model) in [
6488 (ProviderKind::Xai, "grok-4.7"),
6489 (ProviderKind::Xai, "grok-4.6"),
6490 (ProviderKind::XiaomiMimo, "mimo-v2.7-pro-unreleased"),
6491 (ProviderKind::XiaomiMimo, "mimo-v2.6-pro"),
6492 (ProviderKind::Openai, "qwen3-coder"),
6493 (ProviderKind::Deepseek, "qwen3-coder"),
6494 (ProviderKind::Stepfun, "step-3.5"),
6495 (ProviderKind::Together, "any/model"),
6496 (ProviderKind::Ollama, "gpt-oss:20b"),
6497 (ProviderKind::Codewhale, "grok-4.7"),
6498 ] {
6499 assert!(
6500 is_reasoning_model_for_stream(provider, model),
6501 "{provider:?} {model} must surface reasoning fields as Thinking"
6502 );
6503 }
6504 assert_eq!(
6505 super::reasoning_stream_style_for_stream(ProviderKind::Xai, "grok-4.7", Some("none")),
6506 ReasoningStreamStyle::None,
6507 "an explicit route override still wins"
6508 );
6509 }
6510
6511 #[test]
6512 fn xiaomi_mimo_replays_reasoning_for_every_model_id() {
6513 // #6501: MiMo returns 400 when a tool-call turn omits the assistant
6514 // `reasoning_content`, including ids newer than the offline catalog.
6515 for model in ["mimo-v2.5-pro", "mimo-v2.6-pro", "mimo-v2.7-pro-unreleased"] {
6516 assert!(
6517 should_replay_reasoning_content_for_provider(ProviderKind::XiaomiMimo, model, None),
6518 "{model}"
6519 );
6520 }
6521 assert!(!should_replay_reasoning_content_for_provider(
6522 ProviderKind::XiaomiMimo,
6523 "mimo-v2.6-pro",
6524 Some("off"),
6525 ));
6526 }
6527
6528 #[test]
6529 fn stream_display_does_not_authorize_reasoning_replay() {
6530 // #1542 stays fixed: rendering a reasoning delta as Thinking must not
6531 // make a provider that rejects `reasoning_content` receive it back.
6532 for (provider, model) in [
6533 (ProviderKind::Openai, "qwen3-coder"),
6534 (ProviderKind::Openai, "claude-sonnet-4-6"),
6535 (ProviderKind::Xai, "grok-4.7"),
6536 ] {
6537 assert!(is_reasoning_model_for_stream(provider, model));
6538 assert!(
6539 !should_replay_reasoning_content_for_provider(provider, model, None),
6540 "{provider:?} {model}"
6541 );
6542 }
6543 }
6544 }
6545
6546 #[cfg(test)]
6547 mod image_block_wire_tests {
6548 //! The OpenAI-compatible projection of [`ContentBlock::ImageUrl`].
6549 //!
6550 //! Chat Completions is the wire format behind the large majority of
6551 //! CodeWhale's provider routes, so a regression here is a regression for
6552 //! most of them at once. The shape is fixed by OpenAI's spec: a `user`
6553 //! message whose `content` is an array of parts, with the image as
6554 //! `{"type":"image_url","image_url":{"url":…}}`.
6555 use super::{ProviderKind, build_chat_messages, build_chat_wire_body};
6556 use codewhale_models::Role;
6557 use codewhale_models::{ContentBlock, ImageUrlContent, Message, MessageRequest};
6558
6559 const DATA_URL: &str = "data:image/png;base64,QUJD";
6560
6561 #[test]
6562 fn compaction_checkpoint_keeps_complete_tool_round_on_wire() {
6563 let mut messages: Vec<Message> = serde_json::from_value(serde_json::json!([
6564 {"role":"user","content":[{"type":"text","text":"Analyze the data"}]},
6565 {"role":"assistant","content":[{"type":"tool_use","id":"call_1","name":"read","input":{"path":"a.txt"}}]},
6566 {"role":"user","content":[{"type":"tool_result","tool_use_id":"call_1","content":"ready"}]},
6567 {"role":"assistant","content":[{"type":"tool_use","id":"call_2","name":"read","input":{"path":"b.txt"}}]},
6568 {"role":"user","content":[{"type":"tool_result","tool_use_id":"call_2","content":"done"}]}
6569 ])).unwrap();
6570 let summary = codewhale_models::SystemPrompt::Text(
6571 crate::compaction::build_compaction_summary_block_text("Compacted summary", ""),
6572 );
6573 messages.push(crate::compaction::compaction_checkpoint_message(&summary));
6574 crate::runtime_handoff::replace_agent_topology_checkpoint(&mut messages, &[]);
6575 let stored = messages.clone();
6576 let wire = build_chat_messages(None, &messages, "gpt-4o");
6577 assert_eq!(
6578 messages, stored,
6579 "request construction must not alter saved history"
6580 );
6581 let roles: Vec<&str> = wire
6582 .iter()
6583 .map(|message| message["role"].as_str().unwrap())
6584 .collect();
6585 assert_eq!(roles, ["user", "assistant", "tool", "assistant", "tool"]);
6586 let prompt = wire[0]["content"].as_str().unwrap();
6587 assert!(prompt.contains("Compacted summary"));
6588 assert!(prompt.contains("Analyze the data"));
6589 assert!(prompt.contains("codewhale.agent_topology.v1"));
6590 assert_eq!(wire[1]["tool_calls"][0]["id"], "call_1");
6591 assert_eq!(wire[2]["tool_call_id"], "call_1");
6592 assert_eq!(wire[3]["tool_calls"][0]["id"], "call_2");
6593 assert_eq!(wire[4]["tool_call_id"], "call_2");
6594
6595 messages.push(
6596 serde_json::from_value(serde_json::json!({
6597 "role":"assistant","content":[{"type":"text","text":"Analysis complete"}]
6598 }))
6599 .unwrap(),
6600 );
6601 messages.push(
6602 serde_json::from_value(serde_json::json!({
6603 "role":"user","content":[{"type":"text","text":"What happened next?"}]
6604 }))
6605 .unwrap(),
6606 );
6607 let later_wire = build_chat_messages(None, &messages, "gpt-4o");
6608 let later_roles: Vec<&str> = later_wire
6609 .iter()
6610 .map(|message| message["role"].as_str().unwrap())
6611 .collect();
6612 assert_eq!(
6613 later_roles,
6614 [
6615 "user",
6616 "assistant",
6617 "tool",
6618 "assistant",
6619 "tool",
6620 "assistant",
6621 "user"
6622 ]
6623 );
6624 assert_eq!(later_wire[6]["content"], "What happened next?");
6625
6626 let restored = crate::compaction::restore_compaction_checkpoint(
6627 crate::runtime_handoff::project_owned_messages_for_restore(messages.clone()),
6628 Some(&summary),
6629 );
6630 let restored_wire = build_chat_messages(None, &restored, "gpt-4o");
6631 let restored_roles: Vec<&str> = restored_wire
6632 .iter()
6633 .map(|message| message["role"].as_str().unwrap())
6634 .collect();
6635 assert_eq!(restored_roles, later_roles);
6636 assert!(
6637 restored_wire[0]["content"]
6638 .as_str()
6639 .unwrap()
6640 .contains("restored Agent topology checkpoint")
6641 );
6642 assert_eq!(restored_wire[6]["content"], "What happened next?");
6643 }
6644
6645 #[test]
6646 fn quoted_compaction_marker_does_not_reorder_user_wire_messages() {
6647 // The current marker, and the legacy one older sessions still carry.
6648 for marker in [
6649 crate::compaction::COMPACTION_SUMMARY_MARKER,
6650 crate::compaction::LEGACY_V2_COMPACTION_SUMMARY_MARKER,
6651 ] {
6652 let quote = format!("Please explain: {marker}");
6653 let messages: Vec<Message> = serde_json::from_value(serde_json::json!([
6654 {"role":"user","content":[{"type":"text","text":"First question"}]},
6655 {"role":"assistant","content":[{"type":"text","text":"First answer"}]},
6656 {"role":"user","content":[{"type":"text","text":quote}]},
6657 {"role":"assistant","content":[{"type":"text","text":"It introduces a summary."}]},
6658 {"role":"user","content":[{"type":"text","text":"Follow-up question"}]}
6659 ]))
6660 .unwrap();
6661 let wire = build_chat_messages(None, &messages, "gpt-4o");
6662 let roles: Vec<&str> = wire
6663 .iter()
6664 .map(|message| message["role"].as_str().unwrap())
6665 .collect();
6666 assert_eq!(roles, ["user", "assistant", "user", "assistant", "user"]);
6667 assert_eq!(wire[0]["content"], "First question");
6668 assert_eq!(wire[2]["content"], quote);
6669 assert_eq!(wire[4]["content"], "Follow-up question");
6670 }
6671 let messages: Vec<Message> = serde_json::from_value(serde_json::json!([
6672 {"role":"user","content":[{"type":"text","text":"First question"}]},
6673 {"role":"assistant","content":[{"type":"text","text":"First answer"}]},
6674 {"role":"user","content":[{"type":"text","text":"placeholder"}]},
6675 {"role":"assistant","content":[{"type":"text","text":"It introduces a summary."}]},
6676 {"role":"user","content":[{"type":"text","text":"Follow-up question"}]}
6677 ]))
6678 .unwrap();
6679 let roles = ["user", "assistant", "user", "assistant", "user"];
6680
6681 let quoted_exact_header = crate::compaction::build_compaction_summary_block_text(
6682 "This text was pasted by a user",
6683 "",
6684 );
6685 let mut with_exact_quote = messages.clone();
6686 with_exact_quote[2] = Message {
6687 role: Role::User,
6688 content: vec![ContentBlock::Text {
6689 text: quoted_exact_header.clone(),
6690 cache_control: None,
6691 }],
6692 };
6693 let exact_wire = build_chat_messages(None, &with_exact_quote, "gpt-4o");
6694 let exact_roles: Vec<&str> = exact_wire
6695 .iter()
6696 .map(|message| message["role"].as_str().unwrap())
6697 .collect();
6698 assert_eq!(exact_roles, roles);
6699 assert_eq!(exact_wire[2]["content"], quoted_exact_header);
6700
6701 let mut after_tool: Vec<Message> = serde_json::from_value(serde_json::json!([
6702 {"role":"user","content":[{"type":"text","text":"Read first"}]},
6703 {"role":"assistant","content":[{"type":"tool_use","id":"call_1","name":"read","input":{"path":"a.txt"}}]},
6704 {"role":"user","content":[{"type":"tool_result","tool_use_id":"call_1","content":"contents"}]},
6705 {"role":"user","content":[{"type":"text","text":"placeholder"}]},
6706 {"role":"assistant","content":[{"type":"text","text":"Answer"}]}
6707 ])).unwrap();
6708 after_tool[3] = Message {
6709 role: Role::User,
6710 content: vec![ContentBlock::Text {
6711 text: quoted_exact_header.clone(),
6712 cache_control: None,
6713 }],
6714 };
6715 let after_tool_wire = build_chat_messages(None, &after_tool, "gpt-4o");
6716 let after_tool_roles: Vec<&str> = after_tool_wire
6717 .iter()
6718 .map(|message| message["role"].as_str().unwrap())
6719 .collect();
6720 assert_eq!(
6721 after_tool_roles,
6722 ["user", "assistant", "tool", "user", "assistant"]
6723 );
6724 assert_eq!(after_tool_wire[3]["content"], quoted_exact_header);
6725 }
6726
6727 #[test]
6728 fn topology_checkpoint_after_tool_result_keeps_wire_tool_pair() {
6729 let mut messages: Vec<Message> = serde_json::from_value(serde_json::json!([
6730 {"role":"user","content":[{"type":"text","text":"Read the file"}]},
6731 {"role":"assistant","content":[{"type":"tool_use","id":"call_1","name":"read","input":{"path":"a.txt"}}]},
6732 {"role":"user","content":[{"type":"tool_result","tool_use_id":"call_1","content":"contents"}]}
6733 ])).unwrap();
6734 crate::runtime_handoff::replace_agent_topology_checkpoint(&mut messages, &[]);
6735 let wire = build_chat_messages(None, &messages, "gpt-4o");
6736 let roles: Vec<&str> = wire
6737 .iter()
6738 .map(|message| message["role"].as_str().unwrap())
6739 .collect();
6740 assert_eq!(roles, ["user", "assistant", "tool"]);
6741 assert!(
6742 wire[0]["content"]
6743 .as_str()
6744 .unwrap()
6745 .contains("agent_topology_v1")
6746 );
6747 assert_eq!(wire[1]["tool_calls"][0]["id"], "call_1");
6748 assert_eq!(wire[2]["tool_call_id"], "call_1");
6749 }
6750
6751 #[test]
6752 fn compaction_without_retained_user_has_one_wire_user_before_tools() {
6753 let mut messages: Vec<Message> = serde_json::from_value(serde_json::json!([
6754 {"role":"assistant","content":[{"type":"tool_use","id":"call_1","name":"read","input":{"path":"a.txt"}}]},
6755 {"role":"user","content":[{"type":"tool_result","tool_use_id":"call_1","content":"contents"}]}
6756 ])).unwrap();
6757 let summary = codewhale_models::SystemPrompt::Text(
6758 crate::compaction::build_compaction_summary_block_text("Summary", ""),
6759 );
6760 messages.push(crate::compaction::compaction_checkpoint_message(&summary));
6761 crate::runtime_handoff::replace_agent_topology_checkpoint(&mut messages, &[]);
6762 let wire = build_chat_messages(None, &messages, "gpt-4o");
6763 let roles: Vec<&str> = wire
6764 .iter()
6765 .map(|message| message["role"].as_str().unwrap())
6766 .collect();
6767 assert_eq!(roles, ["user", "assistant", "tool"]);
6768 let prompt = wire[0]["content"].as_str().unwrap();
6769 assert!(prompt.contains("Summary"));
6770 assert!(prompt.contains("agent_topology_v1"));
6771 assert_eq!(wire[2]["tool_call_id"], "call_1");
6772 }
6773
6774 #[test]
6775 fn compaction_after_unanswered_user_prompt_has_one_wire_user() {
6776 let mut messages: Vec<Message> = serde_json::from_value(serde_json::json!([
6777 {"role":"user","content":[{"type":"text","text":"Please continue"}]}
6778 ]))
6779 .unwrap();
6780 let summary = codewhale_models::SystemPrompt::Text(
6781 crate::compaction::build_compaction_summary_block_text("Earlier work", ""),
6782 );
6783 messages.push(crate::compaction::compaction_checkpoint_message(&summary));
6784 crate::runtime_handoff::replace_agent_topology_checkpoint(&mut messages, &[]);
6785 let wire = build_chat_messages(None, &messages, "gpt-4o");
6786 assert_eq!(wire.len(), 1);
6787 assert_eq!(wire[0]["role"], "user");
6788 let prompt = wire[0]["content"].as_str().unwrap();
6789 assert!(prompt.contains("Please continue"));
6790 assert!(prompt.contains("Earlier work"));
6791 assert!(prompt.contains("agent_topology_v1"));
6792 }
6793
6794 fn fixture_tool_use(id: &str) -> ContentBlock {
6795 ContentBlock::ToolUse {
6796 execution_id: None,
6797 id: id.to_string(),
6798 name: "read".to_string(),
6799 input: serde_json::json!({"path": format!("{id}.txt")}),
6800 caller: None,
6801 thought_signature: None,
6802 }
6803 }
6804
6805 fn fixture_tool_result(id: &str) -> Message {
6806 Message {
6807 role: Role::User,
6808 content: vec![ContentBlock::ToolResult {
6809 execution_id: None,
6810 tool_use_id: id.to_string(),
6811 content: format!("result for {id}"),
6812 is_error: Some(false),
6813 content_blocks: None,
6814 }],
6815 }
6816 }
6817
6818 fn request_with_image() -> MessageRequest {
6819 MessageRequest {
6820 model: "gpt-4o".to_string(),
6821 messages: vec![Message {
6822 role: Role::User,
6823 content: vec![
6824 ContentBlock::Text {
6825 text: "what is in this screenshot?".to_string(),
6826 cache_control: None,
6827 },
6828 ContentBlock::ImageUrl {
6829 image_url: ImageUrlContent {
6830 url: DATA_URL.to_string(),
6831 },
6832 },
6833 ],
6834 }],
6835 max_tokens: 128,
6836 system: None,
6837 tools: None,
6838 tool_choice: None,
6839 metadata: None,
6840 thinking: None,
6841 reasoning_effort: None,
6842 stream: None,
6843 temperature: None,
6844 top_p: None,
6845 }
6846 }
6847
6848 #[test]
6849 fn user_image_becomes_a_multimodal_parts_array() {
6850 let body = build_chat_wire_body(
6851 &request_with_image(),
6852 ProviderKind::Openai,
6853 "https://api.openai.com/v1",
6854 false,
6855 None,
6856 )
6857 .expect("wire body");
6858
6859 let messages = body.body["messages"].as_array().expect("messages");
6860 let user = messages
6861 .iter()
6862 .find(|message| message["role"] == "user")
6863 .expect("a user message");
6864 let parts = user["content"]
6865 .as_array()
6866 .expect("content must be a parts array once an image is present, not a bare string");
6867
6868 let image = parts
6869 .iter()
6870 .find(|part| part["type"] == "image_url")
6871 .expect("an image_url part");
6872 assert_eq!(image["image_url"]["url"], DATA_URL);
6873
6874 let text = parts
6875 .iter()
6876 .find(|part| part["type"] == "text")
6877 .expect("the accompanying text part");
6878 assert!(
6879 text["text"]
6880 .as_str()
6881 .expect("text")
6882 .contains("what is in this screenshot?"),
6883 "the question must survive alongside the image: {user}"
6884 );
6885 }
6886
6887 #[test]
6888 fn deepseek_vision_exp_uses_chat_image_url_request_shape() {
6889 let mut request = request_with_image();
6890 request.model = "deepseek-v4-flash-vision-exp".to_string();
6891
6892 let body = build_chat_wire_body(
6893 &request,
6894 ProviderKind::Deepseek,
6895 "https://api.deepseek.com/beta",
6896 false,
6897 None,
6898 )
6899 .expect("DeepSeek vision wire body");
6900
6901 assert_eq!(body.body["model"], "deepseek-v4-flash-vision-exp");
6902 let messages = body.body["messages"].as_array().expect("messages");
6903 let user = messages
6904 .iter()
6905 .find(|message| message["role"] == "user")
6906 .expect("a user message");
6907 let parts = user["content"]
6908 .as_array()
6909 .expect("DeepSeek vision content must use multimodal parts");
6910
6911 assert!(parts.iter().any(|part| {
6912 part["type"] == "text" && part["text"] == "what is in this screenshot?"
6913 }));
6914 assert!(
6915 parts.iter().any(|part| {
6916 part["type"] == "image_url" && part["image_url"]["url"] == DATA_URL
6917 })
6918 );
6919 }
6920
6921 #[test]
6922 fn a_message_with_no_image_keeps_its_plain_string_content() {
6923 // Promoting every user turn to a parts array would change the request
6924 // bytes for every text-only route, and with them the prompt-cache
6925 // prefix. Images must be the only thing that triggers the array form.
6926 let mut request = request_with_image();
6927 request.messages[0]
6928 .content
6929 .retain(|block| !matches!(block, ContentBlock::ImageUrl { .. }));
6930
6931 let body = build_chat_wire_body(
6932 &request,
6933 ProviderKind::Openai,
6934 "https://api.openai.com/v1",
6935 false,
6936 None,
6937 )
6938 .expect("wire body");
6939
6940 let messages = body.body["messages"].as_array().expect("messages");
6941 let user = messages
6942 .iter()
6943 .find(|message| message["role"] == "user")
6944 .expect("a user message");
6945 assert!(
6946 user["content"].is_string(),
6947 "text-only turns must stay a plain string: {user}"
6948 );
6949 }
6950
6951 #[test]
6952 fn tool_result_image_follows_its_tool_message_as_multimodal_user_content() {
6953 let mut request = request_with_image();
6954 request.messages = vec![
6955 Message {
6956 role: Role::Assistant,
6957 content: vec![ContentBlock::ToolUse {
6958 execution_id: None,
6959 id: "call_image_1".to_string(),
6960 name: "read".to_string(),
6961 input: serde_json::json!({"path": "shot.png"}),
6962 caller: None,
6963 thought_signature: None,
6964 }],
6965 },
6966 Message {
6967 role: Role::User,
6968 content: vec![ContentBlock::ToolResult {
6969 execution_id: None,
6970 tool_use_id: "call_image_1".to_string(),
6971 content: "screenshot captured".to_string(),
6972 is_error: Some(false),
6973 content_blocks: Some(vec![serde_json::json!({
6974 "type": "image",
6975 "mime_type": "image/png",
6976 "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg==",
6977 })]),
6978 }],
6979 },
6980 ];
6981
6982 let body = build_chat_wire_body(
6983 &request,
6984 ProviderKind::Openai,
6985 "https://api.openai.com/v1",
6986 false,
6987 None,
6988 )
6989 .expect("wire body");
6990 let messages = body.body["messages"].as_array().expect("messages");
6991 let tool_index = messages
6992 .iter()
6993 .position(|message| message["role"] == "tool")
6994 .expect("tool result");
6995 assert_eq!(messages[tool_index]["tool_call_id"], "call_image_1");
6996 assert_eq!(messages[tool_index]["content"], "screenshot captured");
6997
6998 let image_message = &messages[tool_index + 1];
6999 assert_eq!(image_message["role"], "user");
7000 let parts = image_message["content"].as_array().expect("image parts");
7001 assert_eq!(parts[1]["type"], "image_url");
7002 assert_eq!(
7003 parts[1]["image_url"]["url"],
7004 "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg=="
7005 );
7006 assert!(
7007 parts[0]["text"]
7008 .as_str()
7009 .is_some_and(|text| text.contains("read") && text.contains("call_image_1"))
7010 );
7011 }
7012
7013 #[test]
7014 fn tool_result_images_follow_the_entire_tool_call_batch() {
7015 let mut request = request_with_image();
7016 request.messages = vec![
7017 Message {
7018 role: Role::Assistant,
7019 content: vec![
7020 ContentBlock::ToolUse {
7021 execution_id: None,
7022 id: "call_image_1".to_string(),
7023 name: "read".to_string(),
7024 input: serde_json::json!({"path": "first.png"}),
7025 caller: None,
7026 thought_signature: None,
7027 },
7028 ContentBlock::ToolUse {
7029 execution_id: None,
7030 id: "call_image_2".to_string(),
7031 name: "read".to_string(),
7032 input: serde_json::json!({"path": "second.png"}),
7033 caller: None,
7034 thought_signature: None,
7035 },
7036 ],
7037 },
7038 Message {
7039 role: Role::User,
7040 content: vec![ContentBlock::ToolResult {
7041 execution_id: None,
7042 tool_use_id: "call_image_1".to_string(),
7043 content: "first screenshot captured".to_string(),
7044 is_error: Some(false),
7045 content_blocks: Some(vec![serde_json::json!({
7046 "type": "image",
7047 "mime_type": "image/png",
7048 "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg==",
7049 })]),
7050 }],
7051 },
7052 Message {
7053 role: Role::User,
7054 content: vec![ContentBlock::ToolResult {
7055 execution_id: None,
7056 tool_use_id: "call_image_2".to_string(),
7057 content: "second screenshot captured".to_string(),
7058 is_error: Some(false),
7059 content_blocks: Some(vec![serde_json::json!({
7060 "type": "image",
7061 "mime_type": "image/png",
7062 "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGNgYPj/HwADAgH/5ncLrgAAAABJRU5ErkJggg==",
7063 })]),
7064 }],
7065 },
7066 ];
7067
7068 let body = build_chat_wire_body(
7069 &request,
7070 ProviderKind::Openai,
7071 "https://api.openai.com/v1",
7072 false,
7073 None,
7074 )
7075 .expect("wire body");
7076 let messages = body.body["messages"].as_array().expect("messages");
7077 let roles: Vec<_> = messages
7078 .iter()
7079 .map(|message| message["role"].as_str().expect("role"))
7080 .collect();
7081 assert_eq!(roles, ["assistant", "tool", "tool", "user"]);
7082 assert_eq!(messages[1]["tool_call_id"], "call_image_1");
7083 assert_eq!(messages[2]["tool_call_id"], "call_image_2");
7084
7085 let image_parts = messages[3]["content"].as_array().expect("image parts");
7086 assert_eq!(image_parts.len(), 4);
7087 assert_eq!(
7088 image_parts[1]["image_url"]["url"],
7089 "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg=="
7090 );
7091 assert_eq!(
7092 image_parts[3]["image_url"]["url"],
7093 "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGNgYPj/HwADAgH/5ncLrgAAAABJRU5ErkJggg=="
7094 );
7095 }
7096
7097 #[test]
7098 fn out_of_order_tool_results_remain_a_contiguous_complete_batch() {
7099 let messages = vec![
7100 Message {
7101 role: Role::Assistant,
7102 content: vec![fixture_tool_use("call_one"), fixture_tool_use("call_two")],
7103 },
7104 fixture_tool_result("call_two"),
7105 fixture_tool_result("call_one"),
7106 ];
7107
7108 let wire = build_chat_messages(None, &messages, "gpt-4o");
7109 let roles: Vec<_> = wire
7110 .iter()
7111 .map(|message| message["role"].as_str().expect("role"))
7112 .collect();
7113 assert_eq!(roles, ["assistant", "tool", "tool"]);
7114 assert_eq!(wire[1]["tool_call_id"], "call_two");
7115 assert_eq!(wire[2]["tool_call_id"], "call_one");
7116 }
7117
7118 #[test]
7119 fn incomplete_tool_result_batch_is_downgraded_before_serialization() {
7120 let messages = vec![
7121 Message {
7122 role: Role::Assistant,
7123 content: vec![fixture_tool_use("call_one"), fixture_tool_use("call_two")],
7124 },
7125 fixture_tool_result("call_one"),
7126 ];
7127
7128 let wire = build_chat_messages(None, &messages, "gpt-4o");
7129 assert!(
7130 !wire
7131 .iter()
7132 .any(|message| message.get("tool_calls").is_some()),
7133 "an incomplete tool batch must not reach the provider: {wire:?}"
7134 );
7135 assert!(
7136 !wire
7137 .iter()
7138 .any(|message| message["role"].as_str() == Some("tool")),
7139 "the partial result must be removed with its incomplete call batch: {wire:?}"
7140 );
7141 }
7142
7143 #[test]
7144 fn interleaved_tool_result_batch_is_downgraded_before_serialization() {
7145 let messages = vec![
7146 Message {
7147 role: Role::Assistant,
7148 content: vec![
7149 ContentBlock::ToolUse {
7150 execution_id: None,
7151 id: "call_one".to_string(),
7152 name: "read".to_string(),
7153 input: serde_json::json!({"path": "first.png"}),
7154 caller: None,
7155 thought_signature: None,
7156 },
7157 ContentBlock::ToolUse {
7158 execution_id: None,
7159 id: "call_two".to_string(),
7160 name: "read".to_string(),
7161 input: serde_json::json!({"path": "second.png"}),
7162 caller: None,
7163 thought_signature: None,
7164 },
7165 ],
7166 },
7167 Message {
7168 role: Role::User,
7169 content: vec![ContentBlock::ToolResult {
7170 execution_id: None,
7171 tool_use_id: "call_one".to_string(),
7172 content: "first screenshot captured".to_string(),
7173 is_error: Some(false),
7174 content_blocks: Some(vec![serde_json::json!({
7175 "type": "image",
7176 "mime_type": "image/png",
7177 "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg==",
7178 })]),
7179 }],
7180 },
7181 Message {
7182 role: Role::User,
7183 content: vec![
7184 ContentBlock::Text {
7185 text: "interloper".to_string(),
7186 cache_control: None,
7187 },
7188 ContentBlock::ToolResult {
7189 execution_id: None,
7190 tool_use_id: "call_two".to_string(),
7191 content: "second screenshot captured".to_string(),
7192 is_error: Some(false),
7193 content_blocks: Some(vec![serde_json::json!({
7194 "type": "image",
7195 "mime_type": "image/png",
7196 "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg==",
7197 })]),
7198 },
7199 ],
7200 },
7201 ];
7202
7203 let wire = build_chat_messages(None, &messages, "gpt-4o");
7204 assert!(
7205 !wire
7206 .iter()
7207 .any(|message| message.get("tool_calls").is_some()),
7208 "a non-contiguous tool batch must not reach the provider: {wire:?}"
7209 );
7210 assert!(
7211 !wire
7212 .iter()
7213 .any(|message| message["role"].as_str() == Some("tool")),
7214 "orphaned tool replies must be removed with their stripped call batch: {wire:?}"
7215 );
7216 assert!(
7217 !wire.iter().any(|message| {
7218 message["content"].as_array().is_some_and(|parts| {
7219 parts
7220 .iter()
7221 .any(|part| part["type"].as_str() == Some("image_url"))
7222 })
7223 }),
7224 "images from a stripped tool batch must not survive as user input: {wire:?}"
7225 );
7226 }
7227
7228 #[test]
7229 fn duplicate_tool_call_ids_are_downgraded_before_serialization() {
7230 let messages = vec![
7231 Message {
7232 role: Role::Assistant,
7233 content: vec![
7234 ContentBlock::ToolUse {
7235 execution_id: None,
7236 id: "duplicate".to_string(),
7237 name: "read".to_string(),
7238 input: serde_json::json!({"path": "first.png"}),
7239 caller: None,
7240 thought_signature: None,
7241 },
7242 ContentBlock::ToolUse {
7243 execution_id: None,
7244 id: "duplicate".to_string(),
7245 name: "read".to_string(),
7246 input: serde_json::json!({"path": "second.png"}),
7247 caller: None,
7248 thought_signature: None,
7249 },
7250 ],
7251 },
7252 Message {
7253 role: Role::User,
7254 content: vec![ContentBlock::ToolResult {
7255 execution_id: None,
7256 tool_use_id: "duplicate".to_string(),
7257 content: "one result for two calls".to_string(),
7258 is_error: Some(false),
7259 content_blocks: Some(vec![serde_json::json!({
7260 "type": "image",
7261 "mime_type": "image/png",
7262 "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg==",
7263 })]),
7264 }],
7265 },
7266 ];
7267
7268 let wire = build_chat_messages(None, &messages, "gpt-4o");
7269 assert!(
7270 !wire
7271 .iter()
7272 .any(|message| message.get("tool_calls").is_some()),
7273 "duplicate call IDs cannot satisfy two tool calls: {wire:?}"
7274 );
7275 assert!(
7276 !wire
7277 .iter()
7278 .any(|message| message["role"].as_str() == Some("tool")),
7279 "the ambiguous tool result must be removed with the stripped batch: {wire:?}"
7280 );
7281 assert!(
7282 !wire.iter().any(|message| {
7283 message["content"].as_array().is_some_and(|parts| {
7284 parts
7285 .iter()
7286 .any(|part| part["type"].as_str() == Some("image_url"))
7287 })
7288 }),
7289 "media from an ambiguous duplicate-ID batch must not survive: {wire:?}"
7290 );
7291 }
7292 }
7293
7294 #[cfg(test)]
7295 mod mistral_reasoning_tests {
7296 use super::*;
7297
7298 fn request_with_assistant_thinking_and_tool() -> MessageRequest {
7299 MessageRequest {
7300 model: "mistral-medium-latest".to_string(),
7301 messages: vec![
7302 Message {
7303 role: Role::Assistant,
7304 content: vec![
7305 ContentBlock::Thinking {
7306 thinking: "Inspect the current state before calling the tool."
7307 .to_string(),
7308 signature: None,
7309 state: None,
7310 },
7311 ContentBlock::Text {
7312 text: "I will inspect it now.".to_string(),
7313 cache_control: None,
7314 },
7315 ContentBlock::ToolUse {
7316 execution_id: None,
7317 id: "call-1".to_string(),
7318 name: "read_file".to_string(),
7319 input: json!({"path": "README.md"}),
7320 caller: None,
7321 thought_signature: None,
7322 },
7323 ],
7324 },
7325 Message {
7326 role: Role::User,
7327 content: vec![ContentBlock::ToolResult {
7328 execution_id: None,
7329 tool_use_id: "call-1".to_string(),
7330 content: "contents".to_string(),
7331 is_error: None,
7332 content_blocks: None,
7333 }],
7334 },
7335 ],
7336 max_tokens: 64,
7337 system: None,
7338 tools: None,
7339 tool_choice: None,
7340 metadata: None,
7341 thinking: None,
7342 reasoning_effort: Some("high".to_string()),
7343 stream: None,
7344 temperature: None,
7345 top_p: None,
7346 }
7347 }
7348
7349 #[test]
7350 fn mistral_effort_wire_value_covers_codewhale_tiers() {
7351 assert_eq!(mistral_reasoning_effort_wire_value("off"), Some("none"));
7352 assert_eq!(
7353 mistral_reasoning_effort_wire_value("disabled"),
7354 Some("none")
7355 );
7356 assert_eq!(mistral_reasoning_effort_wire_value("none"), Some("none"));
7357 assert_eq!(mistral_reasoning_effort_wire_value("false"), Some("none"));
7358 assert_eq!(mistral_reasoning_effort_wire_value("high"), Some("high"));
7359 assert_eq!(mistral_reasoning_effort_wire_value("xhigh"), Some("high"));
7360 assert_eq!(mistral_reasoning_effort_wire_value("max"), Some("high"));
7361 assert_eq!(mistral_reasoning_effort_wire_value("ultra"), Some("high"));
7362 assert_eq!(
7363 mistral_reasoning_effort_wire_value("ultracode"),
7364 Some("high")
7365 );
7366 // Intermediate tiers must be omitted so the request falls back to
7367 // Mistral's own default rather than 400 code 3051 on unsupported
7368 // values like "low"/"medium" that the server does not accept today.
7369 assert_eq!(mistral_reasoning_effort_wire_value("low"), None);
7370 assert_eq!(mistral_reasoning_effort_wire_value("medium"), None);
7371 assert_eq!(mistral_reasoning_effort_wire_value("mid"), None);
7372 assert_eq!(mistral_reasoning_effort_wire_value("minimal"), None);
7373 }
7374
7375 #[test]
7376 fn mistral_model_gate_only_matches_reasoning_capable_families() {
7377 assert!(mistral_model_supports_reasoning("mistral-medium-latest"));
7378 assert!(mistral_model_supports_reasoning("mistral-medium-3-5"));
7379 assert!(mistral_model_supports_reasoning("mistral-small-latest"));
7380 assert!(mistral_model_supports_reasoning("mistral-small-2603"));
7381 assert!(mistral_model_supports_reasoning("magistral-small-latest"));
7382 assert!(mistral_model_supports_reasoning("MISTRAL-MEDIUM-LATEST"));
7383 assert!(!mistral_model_supports_reasoning("mistral-code-latest"));
7384 assert!(!mistral_model_supports_reasoning("codestral-latest"));
7385 assert!(!mistral_model_supports_reasoning("mistral-large-latest"));
7386 assert!(!mistral_model_supports_reasoning("mistral-nemo-2407"));
7387 }
7388
7389 #[test]
7390 fn mistral_route_shaper_writes_reasoning_only_for_supported_models() {
7391 let mut body = json!({"model": "mistral-medium-latest"});
7392 apply_mistral_route_reasoning_controls(
7393 &mut body,
7394 ProviderKind::Mistral,
7395 crate::config::DEFAULT_MISTRAL_BASE_URL,
7396 "mistral-medium-latest",
7397 Some("high"),
7398 );
7399 assert_eq!(body["reasoning_effort"], json!("high"));
7400
7401 let mut body = json!({"model": "mistral-code-latest"});
7402 apply_mistral_route_reasoning_controls(
7403 &mut body,
7404 ProviderKind::Mistral,
7405 crate::config::DEFAULT_MISTRAL_BASE_URL,
7406 "mistral-code-latest",
7407 Some("high"),
7408 );
7409 assert!(
7410 body.get("reasoning_effort").is_none(),
7411 "non-reasoning models must never see reasoning_effort (Mistral 400s on 3051): {body}"
7412 );
7413
7414 let mut body = json!({"model": "mistral-medium-latest", "reasoning_effort": "stale"});
7415 apply_mistral_route_reasoning_controls(
7416 &mut body,
7417 ProviderKind::Mistral,
7418 crate::config::DEFAULT_MISTRAL_BASE_URL,
7419 "mistral-medium-latest",
7420 Some("low"),
7421 );
7422 assert!(
7423 body.get("reasoning_effort").is_none(),
7424 "intermediate tiers must be stripped rather than sent unsupported: {body}"
7425 );
7426
7427 // Non-Mistral providers must not be touched by this shaper.
7428 let mut body = json!({"model": "deepseek-v4-pro", "reasoning_effort": "high"});
7429 apply_mistral_route_reasoning_controls(
7430 &mut body,
7431 ProviderKind::Deepseek,
7432 crate::config::DEFAULT_MISTRAL_BASE_URL,
7433 "deepseek-v4-pro",
7434 Some("high"),
7435 );
7436 assert_eq!(body["reasoning_effort"], json!("high"));
7437
7438 let mut body = json!({
7439 "model": "mistral-medium-latest",
7440 "thinking": {"type": "enabled"},
7441 "reasoning_effort": "stale",
7442 });
7443 apply_mistral_route_reasoning_controls(
7444 &mut body,
7445 ProviderKind::Mistral,
7446 "https://gateway.example.test/v1",
7447 "mistral-medium-latest",
7448 Some("high"),
7449 );
7450 assert!(body.get("thinking").is_none());
7451 assert!(body.get("reasoning_effort").is_none());
7452
7453 let mut native = json!({"model": "magistral-small-latest"});
7454 apply_mistral_route_reasoning_controls(
7455 &mut native,
7456 ProviderKind::Mistral,
7457 crate::config::DEFAULT_MISTRAL_BASE_URL,
7458 "magistral-small-latest",
7459 Some("off"),
7460 );
7461 assert!(
7462 native.get("reasoning_effort").is_none(),
7463 "legacy native Magistral is always-reasoning and does not use the adjustable effort field"
7464 );
7465 }
7466
7467 #[test]
7468 fn mistral_wire_dialect_is_limited_to_exact_first_party_routes() {
7469 for official in [
7470 "https://api.mistral.ai/v1",
7471 "https://api.eu.mistral.ai/v1/",
7472 "https://api.us.mistral.ai/v1",
7473 ] {
7474 assert!(is_exact_mistral_chat_route(ProviderKind::Mistral, official));
7475 }
7476 for neighbor in [
7477 "http://api.mistral.ai/v1",
7478 "https://api.mistral.ai/v2",
7479 "https://proxy.example.test/v1",
7480 "https://api.mistral.ai.evil.test/v1",
7481 ] {
7482 assert!(!is_exact_mistral_chat_route(
7483 ProviderKind::Mistral,
7484 neighbor
7485 ));
7486 }
7487 assert!(!is_exact_mistral_chat_route(
7488 ProviderKind::Openai,
7489 crate::config::DEFAULT_MISTRAL_BASE_URL,
7490 ));
7491 assert_eq!(
7492 reasoning_stream_style_for_route(
7493 ProviderKind::Mistral,
7494 crate::config::DEFAULT_MISTRAL_BASE_URL,
7495 "mistral-medium-latest",
7496 None,
7497 ),
7498 ReasoningStreamStyle::MistralBlocks
7499 );
7500 assert_eq!(
7501 reasoning_stream_style_for_route(
7502 ProviderKind::Mistral,
7503 "https://gateway.example.test/v1",
7504 "mistral-medium-latest",
7505 None,
7506 ),
7507 ReasoningStreamStyle::SeparateField
7508 );
7509 }
7510
7511 #[test]
7512 fn extract_mistral_polymorphic_content_flattens_thinking_and_text() {
7513 // Non-reasoning response: plain string content. Extractor returns
7514 // (None, None) so the shared string fallback still runs.
7515 let plain = json!({"content": "Hello world"});
7516 assert_eq!(extract_mistral_polymorphic_content(&plain), (None, None));
7517
7518 // Missing content: no panic, returns (None, None).
7519 let empty = json!({});
7520 assert_eq!(extract_mistral_polymorphic_content(&empty), (None, None));
7521
7522 // Reasoning response: nested thinking array + text block.
7523 let reasoning = json!({"content": [
7524 {"type": "thinking", "thinking": [
7525 {"type": "text", "text": "First "},
7526 {"type": "text", "text": "second."},
7527 ], "closed": true},
7528 {"type": "text", "text": "Final answer."},
7529 ]});
7530 let (thinking, text) = extract_mistral_polymorphic_content(&reasoning);
7531 assert_eq!(thinking.as_deref(), Some("First second."));
7532 assert_eq!(text.as_deref(), Some("Final answer."));
7533
7534 // Thinking-only chunk (mid-stream) with no closing text yet.
7535 let thinking_only = json!({"content": [
7536 {"type": "thinking", "thinking": [{"type": "text", "text": "still thinking"}]},
7537 ]});
7538 let (thinking, text) = extract_mistral_polymorphic_content(&thinking_only);
7539 assert_eq!(thinking.as_deref(), Some("still thinking"));
7540 assert_eq!(text, None);
7541 }
7542
7543 #[test]
7544 fn reshape_mistral_messages_reconstructs_polymorphic_shape_for_assistant_replay() {
7545 // Assistant message with stored reasoning_content is reshaped into
7546 // Mistral's polymorphic content-as-array shape.
7547 let mut messages = vec![
7548 json!({"role": "user", "content": "compute 3+4"}),
7549 json!({
7550 "role": "assistant",
7551 "content": "The answer is 7.",
7552 "reasoning_content": "Let me add 3 and 4 to get 7.",
7553 }),
7554 json!({"role": "user", "content": "now multiply by 2"}),
7555 ];
7556 reshape_mistral_messages_for_reasoning_replay(&mut messages);
7557
7558 assert_eq!(messages[0]["role"], "user");
7559 assert!(
7560 messages[0]["content"].is_string(),
7561 "user turns are left untouched: {}",
7562 messages[0]
7563 );
7564
7565 assert_eq!(messages[1]["role"], "assistant");
7566 assert!(
7567 messages[1].get("reasoning_content").is_none(),
7568 "reasoning_content field must be removed after reshape: {}",
7569 messages[1]
7570 );
7571 let content = messages[1]["content"]
7572 .as_array()
7573 .expect("assistant content is now an array");
7574 assert_eq!(content.len(), 2);
7575 assert_eq!(content[0]["type"], "thinking");
7576 assert_eq!(content[0]["closed"], true);
7577 assert_eq!(content[0]["thinking"][0]["type"], "text");
7578 assert_eq!(
7579 content[0]["thinking"][0]["text"],
7580 "Let me add 3 and 4 to get 7."
7581 );
7582 assert_eq!(content[1]["type"], "text");
7583 assert_eq!(content[1]["text"], "The answer is 7.");
7584
7585 // Assistant with no reasoning stays as-is (plain string content).
7586 let mut plain = vec![json!({"role": "assistant", "content": "hi"})];
7587 reshape_mistral_messages_for_reasoning_replay(&mut plain);
7588 assert_eq!(plain[0]["content"], "hi");
7589
7590 // Empty reasoning is treated as absent — no reshape.
7591 let mut empty = vec![json!({
7592 "role": "assistant",
7593 "content": "hi",
7594 "reasoning_content": " ",
7595 })];
7596 reshape_mistral_messages_for_reasoning_replay(&mut empty);
7597 assert_eq!(empty[0]["content"], "hi");
7598 }
7599
7600 #[test]
7601 fn mistral_prompt_builder_replays_stored_thinking_as_polymorphic_content() {
7602 let request = request_with_assistant_thinking_and_tool();
7603 let exact = build_chat_messages_for_request_and_provider_and_route(
7604 &request,
7605 ProviderKind::Mistral,
7606 crate::config::DEFAULT_MISTRAL_BASE_URL,
7607 );
7608 let assistant = &exact[0];
7609 assert!(assistant.get("reasoning_content").is_none());
7610 assert!(assistant.get("tool_calls").is_some());
7611 let content = assistant["content"]
7612 .as_array()
7613 .expect("exact Mistral history uses polymorphic content");
7614 assert_eq!(content[0]["type"], "thinking");
7615 assert_eq!(
7616 content[0]["thinking"][0]["text"],
7617 "Inspect the current state before calling the tool."
7618 );
7619 assert_eq!(content[1]["type"], "text");
7620 assert_eq!(content[1]["text"], "I will inspect it now.");
7621
7622 for (provider, base_url) in [
7623 (ProviderKind::Mistral, "https://gateway.example.test/v1"),
7624 (
7625 ProviderKind::Openai,
7626 crate::config::DEFAULT_MISTRAL_BASE_URL,
7627 ),
7628 ] {
7629 let neighbor = build_chat_messages_for_request_and_provider_and_route(
7630 &request, provider, base_url,
7631 );
7632 assert!(neighbor[0].get("reasoning_content").is_none());
7633 assert!(
7634 neighbor[0]["content"].is_string(),
7635 "unproven routes must not inherit Mistral's polymorphic dialect: {}",
7636 neighbor[0]
7637 );
7638 }
7639 }
7640
7641 #[test]
7642 fn mistral_stream_tool_call_replay_does_not_gain_reasoning_content() {
7643 let request = request_with_assistant_thinking_and_tool();
7644 let wire = build_chat_wire_body(
7645 &request,
7646 ProviderKind::Mistral,
7647 crate::config::DEFAULT_MISTRAL_BASE_URL,
7648 true,
7649 None,
7650 )
7651 .expect("Mistral stream wire body");
7652 let assistant = &wire.body["messages"][0];
7653 assert!(assistant["content"].is_array());
7654 assert!(assistant.get("reasoning_content").is_none());
7655 assert!(assistant.get("tool_calls").is_some());
7656 assert_eq!(wire.body["reasoning_effort"], "high");
7657 assert_eq!(wire.replay_input_tokens, None);
7658 }
7659
7660 #[test]
7661 fn mistral_nonstream_parser_is_route_isolated() {
7662 let payload = json!({
7663 "id": "chatcmpl-mistral",
7664 "model": "mistral-medium-latest",
7665 "choices": [{
7666 "finish_reason": "stop",
7667 "message": {"role": "assistant", "content": [
7668 {"type": "thinking", "thinking": [
7669 {"type": "text", "text": "private trace"}
7670 ], "closed": true},
7671 {"type": "text", "text": "public answer"}
7672 ]}
7673 }],
7674 "usage": {"prompt_tokens": 5, "completion_tokens": 2}
7675 });
7676 let mistral = parse_chat_message_for_route(
7677 &payload,
7678 ProviderKind::Mistral,
7679 crate::config::DEFAULT_MISTRAL_BASE_URL,
7680 )
7681 .expect("Mistral payload parses");
7682 assert!(matches!(
7683 &mistral.content[0],
7684 ContentBlock::Thinking { thinking, .. } if thinking == "private trace"
7685 ));
7686 assert!(matches!(
7687 &mistral.content[1],
7688 ContentBlock::Text { text, .. } if text == "public answer"
7689 ));
7690
7691 let generic = parse_chat_message(&payload).expect("generic payload parses");
7692 assert!(
7693 !generic
7694 .content
7695 .iter()
7696 .any(|block| matches!(block, ContentBlock::Thinking { .. })),
7697 "typed arrays from another provider must not be reinterpreted as Mistral thinking"
7698 );
7699 }
7700
7701 #[test]
7702 fn mistral_shared_capability_matches_wire_contract() {
7703 for model in [
7704 "mistral-medium-latest",
7705 "mistral-small-latest",
7706 "magistral-small-latest",
7707 ] {
7708 assert!(codewhale_models::model_supports_reasoning(model), "{model}");
7709 }
7710 for model in ["mistral-code-latest", "mistral-large-latest"] {
7711 assert!(
7712 !codewhale_models::model_supports_reasoning(model),
7713 "{model}"
7714 );
7715 }
7716 }
7717 }
7718
7719 #[cfg(test)]
7720 mod google_thought_signature_tests {
7721 use super::*;
7722
7723 // ── Google thought signatures (#v0.9.8 Google backend) ──────────────
7724 use crate::config::{DEFAULT_GOOGLE_BASE_URL, DEFAULT_OPENAI_BASE_URL};
7725
7726 /// One signed tool turn. `paired` false is the restart shape: the process
7727 /// died between the tool call and its result, so the terminal
7728 /// `tool_result` never reached durable history.
7729 fn signed_history(signature: Option<&str>, paired: bool) -> Vec<Message> {
7730 let mut messages = vec![
7731 Message {
7732 role: Role::User,
7733 content: vec![ContentBlock::Text {
7734 text: "Read the config.".to_string(),
7735 cache_control: None,
7736 }],
7737 },
7738 Message {
7739 role: Role::Assistant,
7740 content: vec![
7741 ContentBlock::Text {
7742 text: "Reading now.".to_string(),
7743 cache_control: None,
7744 },
7745 ContentBlock::ToolUse {
7746 execution_id: None,
7747 id: "call-g-1".to_string(),
7748 name: "read".to_string(),
7749 input: json!({"path": "config.toml"}),
7750 caller: None,
7751 thought_signature: signature.map(str::to_string),
7752 },
7753 ],
7754 },
7755 ];
7756 if paired {
7757 messages.push(Message {
7758 role: Role::User,
7759 content: vec![ContentBlock::ToolResult {
7760 execution_id: None,
7761 tool_use_id: "call-g-1".to_string(),
7762 content: "key = \"value\"".to_string(),
7763 is_error: None,
7764 content_blocks: None,
7765 }],
7766 });
7767 }
7768 messages
7769 }
7770
7771 fn request_from(messages: Vec<Message>) -> MessageRequest {
7772 MessageRequest {
7773 model: "gemini-3.1-pro-preview".to_string(),
7774 messages,
7775 max_tokens: 64,
7776 system: None,
7777 tools: None,
7778 tool_choice: None,
7779 metadata: None,
7780 thinking: None,
7781 reasoning_effort: Some("high".to_string()),
7782 stream: None,
7783 temperature: None,
7784 top_p: None,
7785 }
7786 }
7787
7788 fn google_request_with_signed_tool(signature: Option<&str>) -> MessageRequest {
7789 request_from(signed_history(signature, true))
7790 }
7791
7792 #[tokio::test]
7793 async fn gateway_thought_signature_rejection_explains_recovery_after_transport() {
7794 use crate::llm_client::LlmClient;
7795 use wiremock::matchers::{method, path};
7796 use wiremock::{Mock, MockServer, ResponseTemplate};
7797
7798 let _ = rustls::crypto::ring::default_provider().install_default();
7799 // An unsigned replay must reach the gateway: it may manage Google's
7800 // signatures itself. Only an actual rejection warrants recovery advice.
7801 for streaming in [false, true] {
7802 for status in [200, 400] {
7803 let server = MockServer::start().await;
7804 let response = if status == 400 {
7805 ResponseTemplate::new(status).set_body_json(json!({
7806 "error": {
7807 "code": 400,
7808 "message": "Function call is missing a thought_signature in functionCall parts."
7809 }
7810 }))
7811 } else if streaming {
7812 ResponseTemplate::new(status)
7813 .insert_header("content-type", "text/event-stream")
7814 .set_body_string("data: [DONE]\n\n")
7815 } else {
7816 ResponseTemplate::new(status).set_body_json(json!({
7817 "id": "gateway-replay",
7818 "model": "gemini-3.1-pro-preview",
7819 "choices": [{
7820 "message": {"role": "assistant", "content": "Done."},
7821 "finish_reason": "stop"
7822 }]
7823 }))
7824 };
7825 Mock::given(method("POST"))
7826 .and(path("/v1/chat/completions"))
7827 .respond_with(response)
7828 .expect(1)
7829 .mount(&server)
7830 .await;
7831
7832 let mut client = CodewhaleClient::new(&crate::config::Config {
7833 provider: Some("openai".to_string()),
7834 providers: Some(crate::config::ProvidersConfig {
7835 openai: crate::config::ProviderConfig {
7836 api_key: Some("gateway-test-key".to_string()),
7837 base_url: Some(format!("{}/v1", server.uri())),
7838 model: Some("gemini-3.1-pro-preview".to_string()),
7839 ..crate::config::ProviderConfig::default()
7840 },
7841 ..crate::config::ProvidersConfig::default()
7842 }),
7843 ..crate::config::Config::default()
7844 })
7845 .expect("gateway client");
7846 client.isolated_request_state = true;
7847 let request = google_request_with_signed_tool(None);
7848 let result = if streaming {
7849 client.create_message_stream(request).await.map(|_| ())
7850 } else {
7851 client
7852 .create_message_without_response_cache(request)
7853 .await
7854 .map(|_| ())
7855 };
7856 if status == 400 {
7857 let error = result.expect_err("gateway rejects unsigned replay");
7858 let message = error.to_string();
7859 assert!(message.contains("built-in `google` provider"), "{message}");
7860 assert!(message.contains("start a new session"), "{message}");
7861 assert!(matches!(
7862 error.downcast_ref::<crate::llm_client::LlmError>(),
7863 Some(crate::llm_client::LlmError::InvalidRequest { status: 400, .. })
7864 ));
7865 } else {
7866 result.expect("gateway-managed signatures must still work");
7867 }
7868 server.verify().await;
7869 }
7870 }
7871 }
7872
7873 #[test]
7874 fn google_route_round_trips_thought_signatures_on_replayed_tool_calls() {
7875 let request = google_request_with_signed_tool(Some("SIG-abc123"));
7876 let messages = build_chat_messages_for_request_and_provider_and_route(
7877 &request,
7878 ProviderKind::Google,
7879 DEFAULT_GOOGLE_BASE_URL,
7880 );
7881 let assistant = messages
7882 .iter()
7883 .find(|m| m.get("role") == Some(&json!("assistant")))
7884 .expect("assistant replay message");
7885 let signature = assistant
7886 .pointer("/tool_calls/0/extra_content/google/thought_signature")
7887 .and_then(serde_json::Value::as_str);
7888 assert_eq!(signature, Some("SIG-abc123"));
7889 }
7890
7891 #[test]
7892 fn google_route_fails_closed_when_replayed_signature_is_missing() {
7893 let request = google_request_with_signed_tool(None);
7894 let error = build_chat_wire_body(
7895 &request,
7896 ProviderKind::Google,
7897 DEFAULT_GOOGLE_BASE_URL,
7898 false,
7899 None,
7900 )
7901 .err()
7902 .expect("missing signature must fail closed before transport");
7903 assert!(
7904 error.to_string().contains("thought signature"),
7905 "error must name the missing signature: {error}"
7906 );
7907 }
7908
7909 /// Google names the same model `gemini-3-pro` and `models/gemini-3-pro` on
7910 /// this endpoint. The prefixed spelling used to match none of the thinking
7911 /// families, so the model that most needs a signature was treated as one
7912 /// that needs none and the replay reached Google unsigned (#6018).
7913 #[test]
7914 fn google_route_fails_closed_for_a_models_prefixed_thinking_id() {
7915 let mut request = google_request_with_signed_tool(None);
7916 request.model = "models/gemini-3-pro-preview".to_string();
7917 let error = build_chat_wire_body(
7918 &request,
7919 ProviderKind::Google,
7920 DEFAULT_GOOGLE_BASE_URL,
7921 false,
7922 None,
7923 )
7924 .err()
7925 .expect("a models/-prefixed thinking id must fail closed like its bare spelling");
7926 assert!(
7927 error.to_string().contains("thought signature"),
7928 "error must name the missing signature: {error}"
7929 );
7930 }
7931
7932 #[test]
7933 fn google_missing_signature_is_a_warning_not_an_error_for_flash_lite() {
7934 // 2.5 Flash-Lite ships thinking off; Google may legitimately omit
7935 // signatures there, so replay proceeds.
7936 let mut request = google_request_with_signed_tool(None);
7937 request.model = "gemini-2.5-flash-lite".to_string();
7938 build_chat_wire_body(
7939 &request,
7940 ProviderKind::Google,
7941 DEFAULT_GOOGLE_BASE_URL,
7942 false,
7943 None,
7944 )
7945 .expect("flash-lite replay must not require a signature");
7946 }
7947
7948 #[test]
7949 fn non_google_routes_never_see_google_extra_content() {
7950 let request = google_request_with_signed_tool(Some("SIG-abc123"));
7951 let messages = build_chat_messages_for_request_and_provider_and_route(
7952 &request,
7953 ProviderKind::Openai,
7954 DEFAULT_OPENAI_BASE_URL,
7955 );
7956 for message in &messages {
7957 if let Some(tool_calls) = message.get("tool_calls").and_then(|v| v.as_array()) {
7958 for call in tool_calls {
7959 assert!(
7960 call.get("extra_content").is_none(),
7961 "Google-only fields must not leak to other providers"
7962 );
7963 }
7964 }
7965 }
7966 }
7967
7968 #[test]
7969 fn google_neighbor_base_url_does_not_get_google_dialect() {
7970 // A Google provider row pointed at some other gateway must not
7971 // carry signatures or fail closed: the dialect binds to the exact
7972 // official route, not to provider identity alone.
7973 let request = google_request_with_signed_tool(None);
7974 build_chat_wire_body(
7975 &request,
7976 ProviderKind::Google,
7977 "https://gateway.example.com/v1",
7978 false,
7979 None,
7980 )
7981 .expect("non-official Google base URL must not require signatures");
7982 let messages = build_chat_messages_for_request_and_provider_and_route(
7983 &google_request_with_signed_tool(Some("SIG")),
7984 ProviderKind::Google,
7985 "https://gateway.example.com/v1",
7986 );
7987 assert!(
7988 messages
7989 .iter()
7990 .all(|m| m.pointer("/tool_calls/0/extra_content").is_none()),
7991 "signatures must not be sent to a non-Google endpoint"
7992 );
7993 }
7994
7995 /// The manually configured OpenAI-compatible row (#1519,
7996 /// `ProviderKind::Custom`) pointed at Google's OpenAI-compat endpoint is
7997 /// byte-for-byte the same endpoint as the built-in `google` row. C22: it
7998 /// used to fail the `provider == Google` half of the route gate, so its
7999 /// signatures were stripped on replay with no warning and later signed
8000 /// tool turns failed. The gate now binds to the endpoint.
8001 #[test]
8002 fn manually_configured_openai_compatible_google_endpoint_replays_signatures() {
8003 let request = google_request_with_signed_tool(Some("SIG-abc123"));
8004 for base_url in [
8005 DEFAULT_GOOGLE_BASE_URL,
8006 "https://generativelanguage.googleapis.com/v1beta/openai",
8007 "https://GenerativeLanguage.googleapis.com/v1beta/openai/",
8008 ] {
8009 let messages = build_chat_messages_for_request_and_provider_and_route(
8010 &request,
8011 ProviderKind::Custom,
8012 base_url,
8013 );
8014 let assistant = messages
8015 .iter()
8016 .find(|m| m.get("role") == Some(&json!("assistant")))
8017 .expect("assistant replay message");
8018 assert_eq!(
8019 assistant
8020 .pointer("/tool_calls/0/extra_content/google/thought_signature")
8021 .and_then(serde_json::Value::as_str),
8022 Some("SIG-abc123"),
8023 "custom row at Google's endpoint must replay the signature ({base_url})"
8024 );
8025 }
8026 }
8027
8028 /// Signature preservation is a property of the endpoint, never of the
8029 /// reasoning setting: an operator who turns reasoning off (or whose
8030 /// effort is simply absent) still has to replay signed tool history.
8031 #[test]
8032 fn signatures_survive_replay_regardless_of_the_reasoning_setting() {
8033 for effort in [None, Some("off"), Some("low"), Some("high")] {
8034 for (provider, base_url) in [
8035 (ProviderKind::Google, DEFAULT_GOOGLE_BASE_URL),
8036 (
8037 ProviderKind::Custom,
8038 "https://generativelanguage.googleapis.com/v1beta/openai",
8039 ),
8040 ] {
8041 let mut request = google_request_with_signed_tool(Some("SIG-abc123"));
8042 request.reasoning_effort = effort.map(str::to_string);
8043 let body = build_chat_wire_body(&request, provider, base_url, true, None)
8044 .expect("signed replay builds on a signature-bearing route");
8045 let assistant = body.body["messages"]
8046 .as_array()
8047 .expect("messages")
8048 .iter()
8049 .find(|m| m.get("role") == Some(&json!("assistant")))
8050 .expect("assistant replay message");
8051 assert_eq!(
8052 assistant
8053 .pointer("/tool_calls/0/extra_content/google/thought_signature")
8054 .and_then(serde_json::Value::as_str),
8055 Some("SIG-abc123"),
8056 "reasoning={effort:?} must not govern signature replay ({base_url})"
8057 );
8058 }
8059 }
8060 }
8061
8062 /// Fail closed on the manually configured row too — the missing-signature
8063 /// error is the useful feedback that replaces a silent strip.
8064 #[test]
8065 fn custom_row_at_google_endpoint_fails_closed_without_a_signature() {
8066 let request = google_request_with_signed_tool(None);
8067 let error = build_chat_wire_body(
8068 &request,
8069 ProviderKind::Custom,
8070 "https://generativelanguage.googleapis.com/v1beta/openai",
8071 false,
8072 None,
8073 )
8074 .err()
8075 .expect("missing signature must fail closed before transport");
8076 let rendered = error.to_string();
8077 assert!(
8078 rendered.contains("thought signature") && rendered.contains("call-g-1"),
8079 "error must name the missing signature and the tool call: {rendered}"
8080 );
8081 }
8082
8083 /// A custom row pointed somewhere else is still a foreign gateway: the
8084 /// signature is stripped, and the strip reports how much it removed so
8085 /// the drop is never silent.
8086 #[test]
8087 fn custom_row_off_google_endpoint_strips_and_reports_signatures() {
8088 let request = google_request_with_signed_tool(Some("SIG-abc123"));
8089 let mut messages = build_chat_messages_for_request_and_provider_and_route(
8090 &request,
8091 ProviderKind::Google,
8092 DEFAULT_GOOGLE_BASE_URL,
8093 );
8094 assert_eq!(
8095 strip_google_tool_call_extra_content(&mut messages),
8096 1,
8097 "the strip must report the signatures it dropped"
8098 );
8099 assert!(
8100 messages
8101 .iter()
8102 .all(|m| m.pointer("/tool_calls/0/extra_content").is_none()),
8103 "stripped history must carry no Google-only fields"
8104 );
8105 let via_route = build_chat_messages_for_request_and_provider_and_route(
8106 &request,
8107 ProviderKind::Custom,
8108 "https://gateway.example.com/v1",
8109 );
8110 assert!(
8111 via_route
8112 .iter()
8113 .all(|m| m.pointer("/tool_calls/0/extra_content").is_none()),
8114 "signatures must not reach a non-Google endpoint"
8115 );
8116 }
8117
8118 #[test]
8119 fn google_reasoning_uses_compatible_effort_without_rejected_native_fields() {
8120 // Wire examples and model limits from Google's compatibility docs.
8121 // Cover the actual request builder, both transport modes and both
8122 // ways a fresh install can configure the official endpoint (#6018).
8123 for (model, effort, expected) in [
8124 ("gemini-3.1-pro-preview", Some("low"), Some("low")),
8125 ("gemini-3.1-pro-preview", Some("medium"), Some("medium")),
8126 ("gemini-3.1-pro-preview", Some("high"), Some("high")),
8127 ("gemini-3.1-pro-preview", Some("max"), Some("high")),
8128 ("gemini-3.5-flash-lite", Some("off"), Some("minimal")),
8129 ("gemini-2.5-flash", Some("off"), Some("none")),
8130 ("models/gemini-2.5-flash-lite", Some("off"), Some("none")),
8131 ("gemini-2.5-pro", Some("off"), Some("minimal")),
8132 ("gemini-3.1-pro-preview", None, None),
8133 ] {
8134 for provider in [ProviderKind::Google, ProviderKind::Custom] {
8135 for streaming in [false, true] {
8136 let mut request = google_request_with_signed_tool(Some("SIG"));
8137 request.model = model.to_string();
8138 request.reasoning_effort = effort.map(str::to_string);
8139 let wire = build_chat_wire_body(
8140 &request,
8141 provider,
8142 DEFAULT_GOOGLE_BASE_URL,
8143 streaming,
8144 None,
8145 )
8146 .expect("valid signed Google request");
8147 assert_eq!(
8148 wire.body.get("reasoning_effort").and_then(Value::as_str),
8149 expected,
8150 "{model}: {effort:?}, {provider:?}, streaming={streaming}"
8151 );
8152 assert!(wire.body.get("google").is_none());
8153 assert!(wire.body.get("extra_body").is_none());
8154 assert!(wire.body.get("thinking").is_none());
8155 }
8156 }
8157 }
8158 }
8159
8160 #[test]
8161 fn google_reasoning_control_does_not_rewrite_other_endpoints() {
8162 for provider in [ProviderKind::Google, ProviderKind::Custom] {
8163 let request = google_request_with_signed_tool(Some("SIG"));
8164 let wire = build_chat_wire_body(
8165 &request,
8166 provider,
8167 "https://gateway.example.com/v1",
8168 false,
8169 None,
8170 )
8171 .expect("valid gateway request");
8172 assert!(wire.body.get("reasoning_effort").is_none());
8173 assert!(wire.body.get("google").is_none());
8174 assert!(wire.body.get("extra_body").is_none());
8175 }
8176 }
8177
8178 #[test]
8179 fn google_signature_captured_from_non_streaming_tool_call() {
8180 let payload = json!({
8181 "id": "resp-1",
8182 "model": "gemini-3.1-pro-preview",
8183 "choices": [{
8184 "index": 0,
8185 "finish_reason": "tool_calls",
8186 "message": {
8187 "role": "assistant",
8188 "content": null,
8189 "tool_calls": [{
8190 "id": "call-g-9",
8191 "type": "function",
8192 "function": {
8193 "name": "read",
8194 "arguments": "{\"path\":\"x\"}"
8195 },
8196 "extra_content": {
8197 "google": { "thought_signature": "SIG-stream" }
8198 }
8199 }]
8200 }
8201 }],
8202 "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}
8203 });
8204 let response = parse_chat_message(&payload).expect("parses");
8205 let signature = response.content.iter().find_map(|block| match block {
8206 ContentBlock::ToolUse {
8207 thought_signature, ..
8208 } => thought_signature.clone(),
8209 _ => None,
8210 });
8211 assert_eq!(signature.as_deref(), Some("SIG-stream"));
8212 }
8213
8214 #[test]
8215 fn google_signature_captured_from_streaming_first_chunk() {
8216 let chunk = json!({
8217 "choices": [{
8218 "index": 0,
8219 "delta": {
8220 "tool_calls": [{
8221 "index": 0,
8222 "id": "call-g-7",
8223 "type": "function",
8224 "function": { "name": "read", "arguments": "{}" },
8225 "extra_content": {
8226 "google": { "thought_signature": "SIG-delta" }
8227 }
8228 }]
8229 }
8230 }]
8231 });
8232 let mut content_index = 0u32;
8233 let mut text_started = false;
8234 let mut thinking_started = false;
8235 let mut tool_indices = std::collections::HashMap::new();
8236 let mut reasoning_buffers = std::collections::HashMap::new();
8237 let mut inline_tags = InlineReasoningTagState::default();
8238 let events = parse_sse_chunk_with_reasoning_style(
8239 &chunk,
8240 &mut content_index,
8241 &mut text_started,
8242 &mut thinking_started,
8243 &mut tool_indices,
8244 &mut reasoning_buffers,
8245 &mut inline_tags,
8246 ReasoningStreamStyle::None,
8247 );
8248 let signature = events.iter().find_map(|event| match event {
8249 StreamEvent::ContentBlockStart {
8250 content_block:
8251 ContentBlockStart::ToolUse {
8252 thought_signature, ..
8253 },
8254 ..
8255 } => thought_signature.clone(),
8256 _ => None,
8257 });
8258 assert_eq!(signature.as_deref(), Some("SIG-delta"));
8259 }
8260
8261 /// Run the production restart/resume chain over a message history:
8262 /// persist to disk, reload, repair crashed tool pairs, then project for
8263 /// restore exactly as `apply.rs` does before assigning `api_messages`.
8264 /// Returns the recovery receipt, the restored messages, and the raw
8265 /// session JSON as it actually sits on disk.
8266 fn resumed(
8267 messages: &[Message],
8268 ) -> (
8269 crate::session_manager::SessionRecovery,
8270 Vec<Message>,
8271 String,
8272 ) {
8273 let dir = tempfile::tempdir().expect("tempdir");
8274 let manager = crate::session_manager::SessionManager::new(dir.path().join("sessions"))
8275 .expect("session manager");
8276 let session = crate::session_manager::create_saved_session(
8277 messages,
8278 "gemini-3.1-pro-preview",
8279 dir.path(),
8280 0,
8281 None,
8282 );
8283 let id = session.metadata.id.clone();
8284 let path = manager.save_session(&session).expect("save session");
8285 let on_disk = std::fs::read_to_string(&path).expect("read persisted session");
8286 let recovery = manager
8287 .recover_session_for_resume(&id)
8288 .expect("recover session for resume");
8289 let restored = crate::runtime_handoff::project_owned_messages_for_restore(
8290 recovery.session.messages.clone(),
8291 );
8292 (recovery, restored, on_disk)
8293 }
8294
8295 fn replayed_signature(body: &serde_json::Value) -> Option<String> {
8296 body["messages"]
8297 .as_array()
8298 .expect("wire messages")
8299 .iter()
8300 // The crash repair appends a trailing assistant text receipt, so
8301 // find the tool-call message by shape, never by index.
8302 .find(|message| message.get("tool_calls").is_some())
8303 .expect("assistant tool-call message")
8304 .pointer("/tool_calls/0/extra_content/google/thought_signature")
8305 .and_then(serde_json::Value::as_str)
8306 .map(str::to_string)
8307 }
8308
8309 /// C22 done-evidence item 3: restarting and resuming must continue the
8310 /// signed history. This walks the whole persistence chain — durable JSON,
8311 /// reload, restore projection, wire body — because the signature can be
8312 /// lost at any of them, and a serde round-trip alone would prove none of
8313 /// it. Local unit evidence only: it does not exercise a real Gemini call.
8314 #[test]
8315 fn resumed_google_session_replays_signed_tool_calls() {
8316 let original = signed_history(Some("SIG-abc123"), true);
8317 let (recovery, restored, on_disk) = resumed(&original);
8318
8319 assert!(
8320 !recovery.changed,
8321 "a fully paired history needs no repair on resume"
8322 );
8323 assert_eq!(
8324 recovery.session.messages, original,
8325 "reload must return the signed history unchanged"
8326 );
8327 assert_eq!(
8328 restored, original,
8329 "the restore projection must not touch signed tool history"
8330 );
8331 assert!(
8332 on_disk.contains("\"thought_signature\"") && on_disk.contains("SIG-abc123"),
8333 "the signature must reach durable storage, not just live memory"
8334 );
8335
8336 for (provider, base_url) in [
8337 (ProviderKind::Google, DEFAULT_GOOGLE_BASE_URL),
8338 (
8339 ProviderKind::Custom,
8340 "https://generativelanguage.googleapis.com/v1beta/openai",
8341 ),
8342 ] {
8343 let body = build_chat_wire_body(
8344 &request_from(restored.clone()),
8345 provider,
8346 base_url,
8347 true,
8348 None,
8349 )
8350 .expect("a resumed signed history must build, not fail closed");
8351 assert_eq!(
8352 replayed_signature(&body.body).as_deref(),
8353 Some("SIG-abc123"),
8354 "resumed history must still replay the signature ({base_url})"
8355 );
8356 }
8357 }
8358
8359 /// The real restart shape: the process died between the tool call and its
8360 /// result, so resume runs `repair_tool_call_pairs`, which rebuilds every
8361 /// message. That rebuild must not strip `ToolUse` fields — if it did, the
8362 /// repaired history would fail closed on Gemini 3 forever after.
8363 #[test]
8364 fn crash_repaired_resume_keeps_the_signature_on_the_repaired_tool_call() {
8365 let (recovery, restored, on_disk) = resumed(&signed_history(Some("SIG-abc123"), false));
8366
8367 assert!(recovery.changed, "a dangling tool call must be repaired");
8368 assert_eq!(recovery.repaired_call_count, 1);
8369 assert_eq!(recovery.duplicate_result_count, 0);
8370 assert_eq!(recovery.orphan_result_count, 0);
8371 assert!(
8372 on_disk.contains("\"thought_signature\"") && on_disk.contains("SIG-abc123"),
8373 "the signature must reach durable storage, not just live memory"
8374 );
8375 assert!(
8376 restored
8377 .iter()
8378 .any(|message| message.content.iter().any(|block| matches!(
8379 block,
8380 ContentBlock::ToolResult { tool_use_id, content, .. }
8381 if tool_use_id == "call-g-1" && content.contains("crashed_and_repaired")
8382 ))),
8383 "the repair must pair the dangling call with a terminal result"
8384 );
8385
8386 let body = build_chat_wire_body(
8387 &request_from(restored),
8388 ProviderKind::Google,
8389 DEFAULT_GOOGLE_BASE_URL,
8390 true,
8391 None,
8392 )
8393 .expect("a crash-repaired signed history must build, not fail closed");
8394 assert_eq!(
8395 replayed_signature(&body.body).as_deref(),
8396 Some("SIG-abc123"),
8397 "crash repair must not strip the thought signature"
8398 );
8399 }
8400 }
8401
8401 lines RUST