| 1 | //! The single prepared-outbound-request seam shared by production dispatch |
| 2 | //! and `/preview-request` (#1004, #3928). |
| 3 | //! |
| 4 | //! Every **primary agent turn** — `LlmClient::create_message` and |
| 5 | //! `create_message_stream`, in Chat Completions, Anthropic Messages, and |
| 6 | //! OpenAI Responses alike — reaches the wire through |
| 7 | //! [`crate::client::CodewhaleClient::prepare_outbound_request`], which returns a |
| 8 | //! [`PreparedOutboundRequest`]. The transports send it; the preview command |
| 9 | //! describes it. Because there is exactly one builder, a preview cannot |
| 10 | //! report a request different from the one a turn would send. |
| 11 | //! |
| 12 | //! Scope, stated plainly: this is *not* every outbound request Codewhale |
| 13 | //! makes. Chat-dialect translation builds its own small fixed body, and FIM, |
| 14 | //! speech, provider-native search, model listing, and the auto-router |
| 15 | //! classifier are separate calls with separate shapes. They are auxiliary and |
| 16 | //! are not described by the request manifest. See `docs/PREVIEW_REQUEST.md`. |
| 17 | //! |
| 18 | //! Nothing in this module performs I/O, mutates client state, or reads the |
| 19 | //! filesystem. It is safe to call on any thread at any time. |
| 20 | //! |
| 21 | //! The seam concept — prepare the exact outbound body once, then let both the |
| 22 | //! sender and the inspector consume it — is harvested from PR #1099 |
| 23 | //! (`build_sanitized_chat_completion_body`) by TaoMu (GTC2080). The |
| 24 | //! implementation here is written against the current multi-dialect client. |
| 25 | |
| 26 | use serde::Serialize; |
| 27 | use serde_json::Value; |
| 28 | |
| 29 | use codewhale_config::provider::WireFormat; |
| 30 | |
| 31 | use crate::config::ProviderKind; |
| 32 | |
| 33 | /// The wire protocol a prepared request speaks. |
| 34 | /// |
| 35 | /// This is the production dialect set. It is deliberately *not* collapsed to |
| 36 | /// Chat Completions: projecting an Anthropic Messages or Responses turn |
| 37 | /// through the Chat builder would describe a body that is never sent. |
| 38 | #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] |
| 39 | #[serde(rename_all = "kebab-case")] |
| 40 | pub(crate) enum WireDialect { |
| 41 | /// OpenAI-style `POST /chat/completions`. |
| 42 | ChatCompletions, |
| 43 | /// Anthropic-style `POST /v1/messages`. |
| 44 | AnthropicMessages, |
| 45 | /// OpenAI-style `POST /responses`. |
| 46 | OpenAiResponses, |
| 47 | } |
| 48 | |
| 49 | impl WireDialect { |
| 50 | pub(crate) fn from_wire_format(format: WireFormat) -> Self { |
| 51 | match format { |
| 52 | WireFormat::ChatCompletions => Self::ChatCompletions, |
| 53 | WireFormat::AnthropicMessages => Self::AnthropicMessages, |
| 54 | WireFormat::Responses => Self::OpenAiResponses, |
| 55 | } |
| 56 | } |
| 57 | |
| 58 | /// Stable machine label. Used in manifests and tests. |
| 59 | pub(crate) fn as_str(self) -> &'static str { |
| 60 | match self { |
| 61 | Self::ChatCompletions => "chat-completions", |
| 62 | Self::AnthropicMessages => "anthropic-messages", |
| 63 | Self::OpenAiResponses => "openai-responses", |
| 64 | } |
| 65 | } |
| 66 | } |
| 67 | |
| 68 | /// The provider-specific *shape* selected inside a dialect. |
| 69 | /// |
| 70 | /// Two routes can share a dialect and still produce structurally different |
| 71 | /// bodies and different endpoint paths (DeepSeek's strict-tools `/beta` path, |
| 72 | /// Kimi Code's nested `thinking.effort`, the ChatGPT Codex Responses path). |
| 73 | /// Naming the shape keeps the manifest honest about which builder branch ran |
| 74 | /// without exposing the route URL. |
| 75 | #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] |
| 76 | #[serde(rename_all = "kebab-case")] |
| 77 | pub(crate) enum RouteShape { |
| 78 | /// Plain dialect defaults for this provider. |
| 79 | Standard, |
| 80 | /// DeepSeek's `/beta/chat/completions` strict-tools path. |
| 81 | DeepseekBetaStrictTools, |
| 82 | /// The exact Kimi Code membership route (nested `thinking.effort`). |
| 83 | KimiCodeK3, |
| 84 | /// The exact pay-as-you-go Moonshot K3 route (fixed sampling). |
| 85 | DirectMoonshotK3, |
| 86 | /// The ChatGPT plan Responses path; identifier retained for saved receipts. |
| 87 | CodexResponses, |
| 88 | /// OpenCode Zen, whose model route re-resolves the wire model per request. |
| 89 | OpencodeZen, |
| 90 | /// A user-configured custom/compatible endpoint on a standard dialect. |
| 91 | CustomCompatible, |
| 92 | } |
| 93 | |
| 94 | impl RouteShape { |
| 95 | pub(crate) fn as_str(self) -> &'static str { |
| 96 | match self { |
| 97 | Self::Standard => "standard", |
| 98 | Self::DeepseekBetaStrictTools => "deepseek-beta-strict-tools", |
| 99 | Self::KimiCodeK3 => "kimi-code-k3", |
| 100 | Self::DirectMoonshotK3 => "direct-moonshot-k3", |
| 101 | Self::CodexResponses => "codex-responses", |
| 102 | Self::OpencodeZen => "opencode-zen", |
| 103 | Self::CustomCompatible => "custom-compatible", |
| 104 | } |
| 105 | } |
| 106 | } |
| 107 | |
| 108 | /// Which endpoint this request would be POSTed to, as typed facts. |
| 109 | /// |
| 110 | /// `url` is the real, unredacted target: production needs it to send. Every |
| 111 | /// display surface must go through [`super::redact_url_for_display`] rather |
| 112 | /// than printing it, and the manifest only ever publishes the redacted |
| 113 | /// scheme/host and a fingerprint — never the path, which can itself carry a |
| 114 | /// deployment secret. |
| 115 | #[derive(Debug, Clone)] |
| 116 | pub(crate) struct EndpointIdentity { |
| 117 | /// Stable provider id (`ProviderKind::as_str`). |
| 118 | pub(crate) provider_id: String, |
| 119 | /// Human-facing provider name. |
| 120 | pub(crate) provider_display: String, |
| 121 | /// The configured route identity when the user named a custom provider, |
| 122 | /// e.g. a `[providers.<name>]` key. `None` for built-ins. |
| 123 | pub(crate) route_id: Option<String>, |
| 124 | /// Full POST target. Never rendered directly. |
| 125 | pub(crate) url: String, |
| 126 | /// Which builder branch produced the body. |
| 127 | pub(crate) shape: RouteShape, |
| 128 | } |
| 129 | |
| 130 | /// What reasoning controls actually landed on the wire, and what was asked |
| 131 | /// for. |
| 132 | /// |
| 133 | /// The receipt is derived from the finished body, not from the intent that |
| 134 | /// went in: if a route-specific shaper stripped `reasoning_effort` and wrote |
| 135 | /// a nested `thinking.effort` instead, that is what this reports. |
| 136 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 137 | pub(crate) struct ReasoningReceipt { |
| 138 | /// The effort string handed to the builder, if any. |
| 139 | pub(crate) requested_effort: Option<String>, |
| 140 | /// Reasoning-shaping fields present on the finished body, in a stable |
| 141 | /// order. Keys only come from the dialect allowlist below, so no message |
| 142 | /// or prompt content can leak through this field. |
| 143 | pub(crate) wire_controls: Vec<(String, Value)>, |
| 144 | } |
| 145 | |
| 146 | /// A key that only *discloses* reasoning output; it does not ask the route to |
| 147 | /// think. Reporting `include` as a reasoning control would make every Responses |
| 148 | /// turn look like a deliberate thinking request. |
| 149 | const REASONING_DISCLOSURE_ONLY_KEYS: &[&str] = &["include"]; |
| 150 | |
| 151 | impl ReasoningReceipt { |
| 152 | /// Reasoning-control keys, per dialect. Anything not on this list is not |
| 153 | /// a reasoning control and never enters the receipt. |
| 154 | fn control_keys(dialect: WireDialect) -> &'static [&'static str] { |
| 155 | match dialect { |
| 156 | WireDialect::ChatCompletions => &[ |
| 157 | "reasoning_effort", |
| 158 | "thinking", |
| 159 | "think", |
| 160 | "reasoning", |
| 161 | "reasoning_split", |
| 162 | "chat_template_kwargs", |
| 163 | ], |
| 164 | WireDialect::AnthropicMessages => &["thinking", "output_config"], |
| 165 | WireDialect::OpenAiResponses => &["reasoning", "include"], |
| 166 | } |
| 167 | } |
| 168 | |
| 169 | fn from_body(dialect: WireDialect, body: &Value, requested_effort: Option<String>) -> Self { |
| 170 | let mut wire_controls = Vec::new(); |
| 171 | for key in Self::control_keys(dialect) { |
| 172 | if let Some(value) = body.get(*key) { |
| 173 | wire_controls.push(((*key).to_string(), value.clone())); |
| 174 | } |
| 175 | } |
| 176 | Self { |
| 177 | requested_effort, |
| 178 | wire_controls, |
| 179 | } |
| 180 | } |
| 181 | |
| 182 | /// The plain `reasoning_effort` string when the route uses that dialect. |
| 183 | pub(crate) fn wire_effort_string(&self) -> Option<&str> { |
| 184 | self.wire_controls |
| 185 | .iter() |
| 186 | .find(|(key, _)| key == "reasoning_effort") |
| 187 | .and_then(|(_, value)| value.as_str()) |
| 188 | } |
| 189 | |
| 190 | /// The effort **actually on the wire**, with the key path it was read from. |
| 191 | /// |
| 192 | /// Flat `reasoning_effort` is only one of the shapes production emits. The |
| 193 | /// Kimi Code route writes `thinking.effort`, the Responses dialect writes |
| 194 | /// `reasoning.effort`, and the Anthropic dialect writes |
| 195 | /// `output_config.effort`. Reporting only the flat key made every nested |
| 196 | /// route read as "no effort sent", which is exactly backwards: those are |
| 197 | /// the routes that were asked to think hardest. |
| 198 | /// |
| 199 | /// The returned key path is a compile-time constant taken from |
| 200 | /// [`Self::control_keys`], never a key read out of the body, so no |
| 201 | /// provider-shaped field name can reach a manifest surface through it. |
| 202 | pub(crate) fn wire_effort(&self) -> Option<(&'static str, &str)> { |
| 203 | if let Some(effort) = self.wire_effort_string() { |
| 204 | return Some(("reasoning_effort", effort)); |
| 205 | } |
| 206 | for (key, value) in &self.wire_controls { |
| 207 | let Some(effort) = value.get("effort").and_then(Value::as_str) else { |
| 208 | continue; |
| 209 | }; |
| 210 | let path = match key.as_str() { |
| 211 | "thinking" => "thinking.effort", |
| 212 | "reasoning" => "reasoning.effort", |
| 213 | "output_config" => "output_config.effort", |
| 214 | "think" => "think.effort", |
| 215 | "reasoning_split" => "reasoning_split.effort", |
| 216 | "chat_template_kwargs" => "chat_template_kwargs.effort", |
| 217 | _ => continue, |
| 218 | }; |
| 219 | return Some((path, effort)); |
| 220 | } |
| 221 | None |
| 222 | } |
| 223 | |
| 224 | /// True when the body actually asks the route to think. |
| 225 | /// |
| 226 | /// Deliberately *not* "the receipt is non-empty": a Responses body that |
| 227 | /// carries only `include: ["reasoning.encrypted_content"]` is *disclosing* |
| 228 | /// reasoning output, not requesting a tier, and must not be reported as an |
| 229 | /// explicit reasoning selection. |
| 230 | pub(crate) fn controls_reasoning(&self) -> bool { |
| 231 | self.wire_controls |
| 232 | .iter() |
| 233 | .any(|(key, _)| !REASONING_DISCLOSURE_ONLY_KEYS.contains(&key.as_str())) |
| 234 | } |
| 235 | } |
| 236 | |
| 237 | /// Which transport entry point asked for this request. |
| 238 | /// |
| 239 | /// This is *caller* intent, not a wire fact. The OpenAI Responses blocking |
| 240 | /// entry point deliberately opens an SSE stream and folds it into one |
| 241 | /// response, so its body carries `"stream": true` while the caller mode is |
| 242 | /// [`Self::Blocking`]. Reporting the two separately is the only way for a |
| 243 | /// manifest to describe the body exactly and still say which entry point it |
| 244 | /// described. See [`PreparedOutboundRequest::wire_stream_field`]. |
| 245 | #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] |
| 246 | #[serde(rename_all = "kebab-case")] |
| 247 | pub(crate) enum CallerStreamMode { |
| 248 | /// `create_message_stream` — the caller consumes stream events. |
| 249 | Streaming, |
| 250 | /// `create_message` — the caller wants one finished response. |
| 251 | Blocking, |
| 252 | } |
| 253 | |
| 254 | impl CallerStreamMode { |
| 255 | pub(crate) fn from_stream_flag(stream: bool) -> Self { |
| 256 | if stream { |
| 257 | Self::Streaming |
| 258 | } else { |
| 259 | Self::Blocking |
| 260 | } |
| 261 | } |
| 262 | |
| 263 | pub(crate) fn as_str(self) -> &'static str { |
| 264 | match self { |
| 265 | Self::Streaming => "streaming", |
| 266 | Self::Blocking => "blocking", |
| 267 | } |
| 268 | } |
| 269 | } |
| 270 | |
| 271 | /// One fully prepared, not-yet-sent outbound request. |
| 272 | /// |
| 273 | /// Both `CodewhaleClient::create_message*` and `/preview-request` consume this |
| 274 | /// value. Adding a field here is how a new wire fact becomes visible to the |
| 275 | /// preview; there is no second builder to keep in sync. |
| 276 | #[derive(Debug, Clone)] |
| 277 | pub(crate) struct PreparedOutboundRequest { |
| 278 | pub(crate) dialect: WireDialect, |
| 279 | pub(crate) endpoint: EndpointIdentity, |
| 280 | /// The model id literally placed on the wire, after route remapping. |
| 281 | pub(crate) wire_model: String, |
| 282 | /// The final, provider-shaped body. This is the exact JSON that would be |
| 283 | /// serialized and POSTed. |
| 284 | pub(crate) body: Value, |
| 285 | pub(crate) reasoning: ReasoningReceipt, |
| 286 | /// Tokens re-sent because thinking-mode replay substituted |
| 287 | /// `reasoning_content` (Chat streaming only). |
| 288 | pub(crate) replay_input_tokens: Option<u32>, |
| 289 | /// Which transport entry point prepared this request. Never a substitute |
| 290 | /// for [`Self::wire_stream_field`], which is the wire truth. |
| 291 | pub(crate) entrypoint: CallerStreamMode, |
| 292 | /// Wire-normalized tool names omitted by a route-specific compatibility |
| 293 | /// check. This stopgap receipt is intentionally not a projection layer; |
| 294 | /// it only transports a bounded user-visible warning. |
| 295 | pub(crate) omitted_tool_names: Vec<String>, |
| 296 | } |
| 297 | |
| 298 | impl PreparedOutboundRequest { |
| 299 | pub(crate) fn new( |
| 300 | dialect: WireDialect, |
| 301 | endpoint: EndpointIdentity, |
| 302 | wire_model: String, |
| 303 | body: Value, |
| 304 | requested_effort: Option<String>, |
| 305 | replay_input_tokens: Option<u32>, |
| 306 | entrypoint: CallerStreamMode, |
| 307 | ) -> Self { |
| 308 | let reasoning = ReasoningReceipt::from_body(dialect, &body, requested_effort); |
| 309 | Self { |
| 310 | dialect, |
| 311 | endpoint, |
| 312 | wire_model, |
| 313 | body, |
| 314 | reasoning, |
| 315 | replay_input_tokens, |
| 316 | entrypoint, |
| 317 | omitted_tool_names: Vec::new(), |
| 318 | } |
| 319 | } |
| 320 | |
| 321 | #[must_use] |
| 322 | pub(crate) fn with_omitted_tool_names(mut self, names: Vec<String>) -> Self { |
| 323 | self.omitted_tool_names = names; |
| 324 | self |
| 325 | } |
| 326 | |
| 327 | /// The `stream` field **as it appears on the finished body**, or `None` |
| 328 | /// when the body carries no such field at all. |
| 329 | /// |
| 330 | /// This is the wire truth and the only value a manifest may present as |
| 331 | /// "what the request says". It is deliberately not derived from |
| 332 | /// [`Self::entrypoint`]: the Responses blocking path sends |
| 333 | /// `"stream": true` and the Chat blocking path omits the field entirely, |
| 334 | /// so both would be misreported by the caller mode. |
| 335 | pub(crate) fn wire_stream_field(&self) -> Option<bool> { |
| 336 | self.body.get("stream").and_then(Value::as_bool) |
| 337 | } |
| 338 | |
| 339 | /// Canonical serialization of the **complete** final body. |
| 340 | /// |
| 341 | /// `serde_json` is built with `preserve_order` in this crate, so insertion |
| 342 | /// order — not key order — drives `to_string`. Canonicalizing here means |
| 343 | /// the hash is stable across builder orderings while still changing when |
| 344 | /// any value anywhere in the body changes: max-token fields, tool choice, |
| 345 | /// nested reasoning controls, transformed tool schemas, attachment parts, |
| 346 | /// stream options, and every message. |
| 347 | pub(crate) fn canonical_body(&self) -> String { |
| 348 | canonical_json(&self.body) |
| 349 | } |
| 350 | |
| 351 | /// SHA-256 over [`Self::canonical_body`]. Whole-body, not a prefix. |
| 352 | pub(crate) fn body_sha256(&self) -> String { |
| 353 | crate::hashing::sha256_hex(self.canonical_body().as_bytes()) |
| 354 | } |
| 355 | |
| 356 | /// Dialect-aware view of the finished body, for counting and estimation. |
| 357 | pub(crate) fn wire_view(&self) -> WireBodyView<'_> { |
| 358 | WireBodyView::extract(self.dialect, &self.body) |
| 359 | } |
| 360 | |
| 361 | /// Attach the caller's named route identity (a `[providers.<name>]` key, |
| 362 | /// or any other route id the resolved turn plan owns). |
| 363 | #[must_use] |
| 364 | pub(crate) fn with_route_id(mut self, route_id: Option<String>) -> Self { |
| 365 | self.endpoint.route_id = route_id; |
| 366 | self |
| 367 | } |
| 368 | |
| 369 | /// SHA-256 of the full endpoint URL. Lets two previews be compared for |
| 370 | /// "same endpoint?" without either of them printing the path. |
| 371 | pub(crate) fn endpoint_fingerprint(&self) -> String { |
| 372 | crate::hashing::sha256_hex(self.endpoint.url.as_bytes()) |
| 373 | } |
| 374 | |
| 375 | /// A bounded endpoint class that never publishes a remote authority. |
| 376 | /// Custom-provider tenant subdomains can contain credentials, so every |
| 377 | /// non-loopback authority is represented by a short digest. |
| 378 | pub(crate) fn safe_endpoint_host_class(&self) -> String { |
| 379 | let Ok(url) = reqwest::Url::parse(&self.endpoint.url) else { |
| 380 | let digest = crate::hashing::sha256_hex(self.endpoint.url.as_bytes()); |
| 381 | return format!("unparseable sha256:{}", &digest[..12]); |
| 382 | }; |
| 383 | let scheme = match url.scheme() { |
| 384 | "http" => "http", |
| 385 | "https" => "https", |
| 386 | _ => "other", |
| 387 | }; |
| 388 | let host = url.host_str().unwrap_or_default(); |
| 389 | let loopback = host.eq_ignore_ascii_case("localhost") |
| 390 | || host |
| 391 | .parse::<std::net::IpAddr>() |
| 392 | .is_ok_and(|address| address.is_loopback()); |
| 393 | if loopback { |
| 394 | return format!("{scheme} loopback"); |
| 395 | } |
| 396 | let authority = url.port().map_or_else( |
| 397 | || host.to_ascii_lowercase(), |
| 398 | |port| format!("{}:{port}", host.to_ascii_lowercase()), |
| 399 | ); |
| 400 | let digest = crate::hashing::sha256_hex(authority.as_bytes()); |
| 401 | format!("{scheme} remote sha256:{}", &digest[..12]) |
| 402 | } |
| 403 | |
| 404 | /// Output cap literally serialized into the primary request body. |
| 405 | pub(crate) fn wire_output_cap_tokens(&self) -> Option<u64> { |
| 406 | ["max_tokens", "max_completion_tokens", "max_output_tokens"] |
| 407 | .into_iter() |
| 408 | .find_map(|key| self.body.get(key).and_then(Value::as_u64)) |
| 409 | } |
| 410 | } |
| 411 | |
| 412 | /// Canonical JSON: object keys sorted, no insignificant whitespace. |
| 413 | /// |
| 414 | /// Deterministic for a given `Value` regardless of how it was built. |
| 415 | pub(crate) fn canonical_json(value: &Value) -> String { |
| 416 | let mut out = String::new(); |
| 417 | write_canonical(value, &mut out); |
| 418 | out |
| 419 | } |
| 420 | |
| 421 | fn write_canonical(value: &Value, out: &mut String) { |
| 422 | match value { |
| 423 | Value::Object(map) => { |
| 424 | let mut keys: Vec<&String> = map.keys().collect(); |
| 425 | keys.sort_unstable(); |
| 426 | out.push('{'); |
| 427 | for (index, key) in keys.iter().enumerate() { |
| 428 | if index > 0 { |
| 429 | out.push(','); |
| 430 | } |
| 431 | push_json_string(key, out); |
| 432 | out.push(':'); |
| 433 | write_canonical(&map[*key], out); |
| 434 | } |
| 435 | out.push('}'); |
| 436 | } |
| 437 | Value::Array(items) => { |
| 438 | out.push('['); |
| 439 | for (index, item) in items.iter().enumerate() { |
| 440 | if index > 0 { |
| 441 | out.push(','); |
| 442 | } |
| 443 | write_canonical(item, out); |
| 444 | } |
| 445 | out.push(']'); |
| 446 | } |
| 447 | other => out.push_str(&other.to_string()), |
| 448 | } |
| 449 | } |
| 450 | |
| 451 | fn push_json_string(value: &str, out: &mut String) { |
| 452 | out.push_str(&Value::String(value.to_string()).to_string()); |
| 453 | } |
| 454 | |
| 455 | /// Where a given dialect keeps its system text, turn items, and tool schemas. |
| 456 | /// |
| 457 | /// Extraction is by shape, never by guessing: a Responses body has |
| 458 | /// `instructions`/`input`, an Anthropic body has `system`/`messages`, a Chat |
| 459 | /// body carries the system prompt as the first `system`-role message. |
| 460 | /// |
| 461 | /// # The byte accounting sums exactly |
| 462 | /// |
| 463 | /// `system_bytes + tool_schema_bytes + item_bytes + framing_bytes == |
| 464 | /// body_bytes`, where `body_bytes` is the length of |
| 465 | /// [`PreparedOutboundRequest::canonical_body`] — a stable, key-sorted semantic |
| 466 | /// serialization of the body. Production sends the same JSON value, but the |
| 467 | /// transport serializer may preserve a different object-key order. These are |
| 468 | /// therefore canonical JSON sizes, not literal HTTP payload byte counts. This |
| 469 | /// is an accounting decomposition, not a set of four borrowed |
| 470 | /// byte ranges in the JSON buffer: |
| 471 | /// |
| 472 | /// - `system_bytes`, `tool_schema_bytes`, and `item_bytes` are the canonical |
| 473 | /// serializations of their *values* (for Chat, `system_bytes` is the |
| 474 | /// serialized system-role messages, carved out of the `messages` array); |
| 475 | /// - `framing_bytes` is the algebraic remainder after those three canonical |
| 476 | /// value-region sizes. It includes every other top-level field and whatever |
| 477 | /// JSON structure was not already counted inside a selected array value. |
| 478 | /// |
| 479 | /// The earlier shape counted selected values and then serialized a *separate* |
| 480 | /// object for "framing", which double-omitted key names, brackets, and |
| 481 | /// separators and made the parts sum to less than the whole. Framing is now |
| 482 | /// defined as the remainder precisely so that cannot happen again; |
| 483 | /// [`WireBodyView::partition_is_exact`] asserts only the sum identity; the |
| 484 | /// regional names remain attribution estimates over canonical values. |
| 485 | /// |
| 486 | /// `tool_result_bytes` and `attachment_bytes` are deliberately *not* part of |
| 487 | /// the partition: they are subsets of `item_bytes`, reported for attribution. |
| 488 | #[derive(Debug, Default)] |
| 489 | pub(crate) struct WireBodyView<'a> { |
| 490 | /// Canonical byte length of the complete wire body. |
| 491 | pub(crate) body_bytes: usize, |
| 492 | /// Serialized bytes of the system/instructions region. |
| 493 | pub(crate) system_bytes: usize, |
| 494 | /// SHA-256 of the canonicalized system/instructions region — the hash of |
| 495 | /// the prompt this prepared request would actually send. Empty when the |
| 496 | /// request carries no system region. |
| 497 | pub(crate) system_sha256: String, |
| 498 | /// Serialized bytes of the tool-schema region. |
| 499 | pub(crate) tool_schema_bytes: usize, |
| 500 | /// SHA-256 of the canonicalized **wire** tool region: the schemas exactly |
| 501 | /// as the provider receives them, after every dialect transform and |
| 502 | /// strict-mode sanitizer. Empty when the body carries no `tools` field. |
| 503 | pub(crate) tool_schema_sha256: String, |
| 504 | /// Number of tool schemas on the wire. |
| 505 | pub(crate) tool_count: usize, |
| 506 | /// Turn items (messages / input items), excluding the system region. |
| 507 | pub(crate) items: Vec<&'a Value>, |
| 508 | /// Serialized bytes of the turn-item region, including the array's own |
| 509 | /// brackets and separators and excluding any carved-out system messages. |
| 510 | pub(crate) item_bytes: usize, |
| 511 | /// Serialized bytes of tool-result items specifically. Subset of |
| 512 | /// [`Self::item_bytes`]. |
| 513 | pub(crate) tool_result_bytes: usize, |
| 514 | /// Number of attachment (image) parts referenced anywhere in the items. |
| 515 | pub(crate) attachment_count: usize, |
| 516 | /// Serialized bytes of those attachment parts. Subset of |
| 517 | /// [`Self::item_bytes`]. |
| 518 | pub(crate) attachment_bytes: usize, |
| 519 | /// Algebraic remainder after the three canonical value-region sizes. This |
| 520 | /// includes other top-level fields and JSON structure not already counted |
| 521 | /// inside a selected array value. |
| 522 | pub(crate) framing_bytes: usize, |
| 523 | } |
| 524 | |
| 525 | impl<'a> WireBodyView<'a> { |
| 526 | fn extract(dialect: WireDialect, body: &'a Value) -> Self { |
| 527 | let body_bytes = canonical_json(body).len(); |
| 528 | let mut view = Self { |
| 529 | body_bytes, |
| 530 | ..Self::default() |
| 531 | }; |
| 532 | let Some(object) = body.as_object() else { |
| 533 | view.framing_bytes = view.body_bytes; |
| 534 | return view; |
| 535 | }; |
| 536 | |
| 537 | let (system_key, items_key) = match dialect { |
| 538 | WireDialect::ChatCompletions => (None, "messages"), |
| 539 | WireDialect::AnthropicMessages => (Some("system"), "messages"), |
| 540 | WireDialect::OpenAiResponses => (Some("instructions"), "input"), |
| 541 | }; |
| 542 | |
| 543 | // The system region is accumulated as canonical text so it can be |
| 544 | // hashed once, then dropped. The text itself never leaves this scope. |
| 545 | let mut system_region = String::new(); |
| 546 | if let Some(key) = system_key |
| 547 | && let Some(system) = object.get(key) |
| 548 | { |
| 549 | system_region.push_str(&canonical_json(system)); |
| 550 | } |
| 551 | |
| 552 | if let Some(tools) = object.get("tools") { |
| 553 | let canonical_tools = canonical_json(tools); |
| 554 | view.tool_schema_bytes = canonical_tools.len(); |
| 555 | view.tool_schema_sha256 = crate::hashing::sha256_hex(canonical_tools.as_bytes()); |
| 556 | view.tool_count = tools.as_array().map_or(0, |tools| { |
| 557 | tools |
| 558 | .iter() |
| 559 | .map(|tool| { |
| 560 | if tool.get("type").and_then(Value::as_str) == Some("namespace") { |
| 561 | tool.get("tools") |
| 562 | .and_then(Value::as_array) |
| 563 | .map_or(0, Vec::len) |
| 564 | } else { |
| 565 | 1 |
| 566 | } |
| 567 | }) |
| 568 | .sum() |
| 569 | }); |
| 570 | } |
| 571 | |
| 572 | if let Some(items_value) = object.get(items_key) { |
| 573 | // The whole array, brackets and separators included, so the |
| 574 | // accounting can include the canonical array value itself. |
| 575 | let mut item_region_bytes = canonical_json(items_value).len(); |
| 576 | if let Some(items) = items_value.as_array() { |
| 577 | for item in items { |
| 578 | let bytes = canonical_json(item).len(); |
| 579 | // Chat Completions carries the system prompt inline as the |
| 580 | // first system-role message. Account for it as system, not |
| 581 | // as conversation, so cross-dialect numbers stay |
| 582 | // comparable — and subtract it from the item region so the |
| 583 | // two never double-count the same bytes. |
| 584 | if dialect == WireDialect::ChatCompletions |
| 585 | && item.get("role").and_then(Value::as_str) == Some("system") |
| 586 | { |
| 587 | system_region.push_str(&canonical_json(item)); |
| 588 | item_region_bytes = item_region_bytes.saturating_sub(bytes); |
| 589 | continue; |
| 590 | } |
| 591 | if is_tool_result_item(dialect, item) { |
| 592 | view.tool_result_bytes = view.tool_result_bytes.saturating_add(bytes); |
| 593 | } |
| 594 | let (count, attachment_bytes) = count_attachments(dialect, item); |
| 595 | view.attachment_count = view.attachment_count.saturating_add(count); |
| 596 | view.attachment_bytes = view.attachment_bytes.saturating_add(attachment_bytes); |
| 597 | view.items.push(item); |
| 598 | } |
| 599 | } |
| 600 | view.item_bytes = item_region_bytes; |
| 601 | } |
| 602 | |
| 603 | view.system_bytes = system_region.len(); |
| 604 | if !system_region.is_empty() { |
| 605 | view.system_sha256 = crate::hashing::sha256_hex(system_region.as_bytes()); |
| 606 | } |
| 607 | |
| 608 | // Framing is the algebraic remainder, never a separately serialized |
| 609 | // object. The sum is exact; these are not four disjoint byte slices. |
| 610 | view.framing_bytes = view |
| 611 | .body_bytes |
| 612 | .saturating_sub(view.system_bytes) |
| 613 | .saturating_sub(view.tool_schema_bytes) |
| 614 | .saturating_sub(view.item_bytes); |
| 615 | view |
| 616 | } |
| 617 | |
| 618 | /// Whether the four partition classes sum to the whole wire body. |
| 619 | /// |
| 620 | /// The manifest publishes these as exact byte facts, so the invariant is |
| 621 | /// asserted in tests across every dialect and both entry points rather |
| 622 | /// than merely documented. |
| 623 | pub(crate) fn partition_is_exact(&self) -> bool { |
| 624 | self.system_bytes |
| 625 | .saturating_add(self.tool_schema_bytes) |
| 626 | .saturating_add(self.item_bytes) |
| 627 | .saturating_add(self.framing_bytes) |
| 628 | == self.body_bytes |
| 629 | } |
| 630 | } |
| 631 | |
| 632 | fn is_tool_result_item(dialect: WireDialect, item: &Value) -> bool { |
| 633 | match dialect { |
| 634 | WireDialect::ChatCompletions => item.get("role").and_then(Value::as_str) == Some("tool"), |
| 635 | WireDialect::AnthropicMessages => item |
| 636 | .get("content") |
| 637 | .and_then(Value::as_array) |
| 638 | .is_some_and(|blocks| { |
| 639 | blocks |
| 640 | .iter() |
| 641 | .any(|block| block.get("type").and_then(Value::as_str) == Some("tool_result")) |
| 642 | }), |
| 643 | WireDialect::OpenAiResponses => { |
| 644 | item.get("type").and_then(Value::as_str) == Some("function_call_output") |
| 645 | } |
| 646 | } |
| 647 | } |
| 648 | |
| 649 | /// Count attachment parts and their serialized size. Only sizes leave this |
| 650 | /// function — never a URL, path, or payload. |
| 651 | fn count_attachments(dialect: WireDialect, item: &Value) -> (usize, usize) { |
| 652 | let Some(parts) = item.get("content").and_then(Value::as_array) else { |
| 653 | return (0, 0); |
| 654 | }; |
| 655 | let mut count = 0usize; |
| 656 | let mut bytes = 0usize; |
| 657 | for part in parts { |
| 658 | let part_type = part.get("type").and_then(Value::as_str); |
| 659 | let is_attachment = match dialect { |
| 660 | WireDialect::ChatCompletions => { |
| 661 | part_type == Some("image_url") || part.get("image_url").is_some() |
| 662 | } |
| 663 | WireDialect::AnthropicMessages => { |
| 664 | matches!(part_type, Some("image" | "document")) |
| 665 | } |
| 666 | WireDialect::OpenAiResponses => { |
| 667 | matches!(part_type, Some("input_image" | "input_file")) |
| 668 | } |
| 669 | }; |
| 670 | if !is_attachment { |
| 671 | continue; |
| 672 | } |
| 673 | count += 1; |
| 674 | bytes = bytes.saturating_add(canonical_json(part).len()); |
| 675 | } |
| 676 | (count, bytes) |
| 677 | } |
| 678 | |
| 679 | /// Classify the provider-specific shape of a prepared Chat Completions body. |
| 680 | pub(crate) fn chat_route_shape( |
| 681 | provider: ProviderKind, |
| 682 | base_url: &str, |
| 683 | wire_model: &str, |
| 684 | url: &str, |
| 685 | ) -> RouteShape { |
| 686 | if provider == ProviderKind::OpencodeZen { |
| 687 | return RouteShape::OpencodeZen; |
| 688 | } |
| 689 | if url.contains("/beta/chat/completions") { |
| 690 | return RouteShape::DeepseekBetaStrictTools; |
| 691 | } |
| 692 | if crate::config::is_exact_kimi_code_k3_route(provider, base_url, wire_model) { |
| 693 | return RouteShape::KimiCodeK3; |
| 694 | } |
| 695 | if crate::config::is_exact_direct_moonshot_k3_route(provider, base_url, wire_model) { |
| 696 | return RouteShape::DirectMoonshotK3; |
| 697 | } |
| 698 | if provider == ProviderKind::Custom { |
| 699 | return RouteShape::CustomCompatible; |
| 700 | } |
| 701 | RouteShape::Standard |
| 702 | } |
| 703 | |
| 704 | #[cfg(test)] |
| 705 | mod tests { |
| 706 | use super::*; |
| 707 | use serde_json::{Map, json}; |
| 708 | |
| 709 | fn endpoint() -> EndpointIdentity { |
| 710 | EndpointIdentity { |
| 711 | provider_id: "deepseek".to_string(), |
| 712 | provider_display: "DeepSeek".to_string(), |
| 713 | route_id: None, |
| 714 | url: "https://api.deepseek.com/chat/completions".to_string(), |
| 715 | shape: RouteShape::Standard, |
| 716 | } |
| 717 | } |
| 718 | |
| 719 | fn prepared(body: Value) -> PreparedOutboundRequest { |
| 720 | PreparedOutboundRequest::new( |
| 721 | WireDialect::ChatCompletions, |
| 722 | endpoint(), |
| 723 | "deepseek-chat".to_string(), |
| 724 | body, |
| 725 | Some("high".to_string()), |
| 726 | None, |
| 727 | CallerStreamMode::Streaming, |
| 728 | ) |
| 729 | } |
| 730 | |
| 731 | #[test] |
| 732 | fn canonical_json_is_key_order_independent() { |
| 733 | let a = json!({"b": 1, "a": {"z": 2, "y": [3, {"q": 4, "p": 5}]}}); |
| 734 | let mut b = Map::new(); |
| 735 | b.insert("a".to_string(), json!({"y": [3, {"p": 5, "q": 4}], "z": 2})); |
| 736 | b.insert("b".to_string(), json!(1)); |
| 737 | assert_eq!(canonical_json(&a), canonical_json(&Value::Object(b))); |
| 738 | assert_eq!( |
| 739 | canonical_json(&a), |
| 740 | r#"{"a":{"y":[3,{"p":5,"q":4}],"z":2},"b":1}"# |
| 741 | ); |
| 742 | } |
| 743 | |
| 744 | #[test] |
| 745 | fn body_hash_covers_every_wire_field() { |
| 746 | let base = prepared(json!({ |
| 747 | "model": "deepseek-chat", |
| 748 | "messages": [{"role": "user", "content": "hi"}], |
| 749 | "max_tokens": 4096, |
| 750 | "tools": [{"type": "function", "function": {"name": "read_file"}}], |
| 751 | "tool_choice": {"type": "auto"}, |
| 752 | "reasoning_effort": "high", |
| 753 | "stream": true, |
| 754 | })); |
| 755 | let baseline = base.body_sha256(); |
| 756 | |
| 757 | // Every one of these is a field a preview would have to notice. |
| 758 | let mutations: Vec<(&str, Value)> = vec![ |
| 759 | ("max_tokens", json!(2048)), |
| 760 | ("tool_choice", json!("required")), |
| 761 | ("reasoning_effort", json!("low")), |
| 762 | ("stream", json!(false)), |
| 763 | ("temperature", json!(0.2)), |
| 764 | ]; |
| 765 | for (key, value) in mutations { |
| 766 | let mut body = base.body.clone(); |
| 767 | body[key] = value; |
| 768 | assert_ne!( |
| 769 | baseline, |
| 770 | prepared(body).body_sha256(), |
| 771 | "mutating `{key}` must change the whole-body hash" |
| 772 | ); |
| 773 | } |
| 774 | |
| 775 | // Nested changes: a transformed tool schema and a nested reasoning |
| 776 | // control both have to move the hash. |
| 777 | let mut nested = base.body.clone(); |
| 778 | nested["tools"][0]["function"]["parameters"] = json!({"type": "object"}); |
| 779 | assert_ne!(baseline, prepared(nested).body_sha256()); |
| 780 | |
| 781 | let mut thinking = base.body.clone(); |
| 782 | thinking["thinking"] = json!({"type": "enabled", "effort": "max"}); |
| 783 | assert_ne!(baseline, prepared(thinking).body_sha256()); |
| 784 | } |
| 785 | |
| 786 | #[test] |
| 787 | fn endpoint_host_class_never_prints_remote_authority_or_path() { |
| 788 | let hostile = |url: &str| { |
| 789 | let mut endpoint = endpoint(); |
| 790 | endpoint.url = url.to_string(); |
| 791 | PreparedOutboundRequest::new( |
| 792 | WireDialect::ChatCompletions, |
| 793 | endpoint, |
| 794 | "model".to_string(), |
| 795 | json!({"model": "model", "messages": []}), |
| 796 | None, |
| 797 | None, |
| 798 | CallerStreamMode::Streaming, |
| 799 | ) |
| 800 | }; |
| 801 | |
| 802 | let token_host = |
| 803 | hostile("https://sk-live-abcdef0123456789.tenant.example/v1/deployments/secret/chat"); |
| 804 | let same_host_other_path = |
| 805 | hostile("https://sk-live-abcdef0123456789.tenant.example/other/private/path"); |
| 806 | let idn = hostile("https://秘密.example/private/path?api_key=secret"); |
| 807 | |
| 808 | let class = token_host.safe_endpoint_host_class(); |
| 809 | assert_eq!(class, same_host_other_path.safe_endpoint_host_class()); |
| 810 | assert_ne!( |
| 811 | token_host.endpoint_fingerprint(), |
| 812 | same_host_other_path.endpoint_fingerprint(), |
| 813 | "the separate full-endpoint fingerprint must still detect path drift" |
| 814 | ); |
| 815 | for forbidden in ["sk-live", "tenant", "example", "deployment", "secret"] { |
| 816 | assert!(!class.contains(forbidden), "{forbidden} leaked in {class}"); |
| 817 | } |
| 818 | let idn_class = idn.safe_endpoint_host_class(); |
| 819 | for forbidden in ["秘密", "xn--", "example", "private", "api_key", "secret"] { |
| 820 | assert!( |
| 821 | !idn_class.contains(forbidden), |
| 822 | "{forbidden} leaked in {idn_class}" |
| 823 | ); |
| 824 | } |
| 825 | assert!(class.starts_with("https remote sha256:"), "{class}"); |
| 826 | assert!(class.len() <= 40, "{class}"); |
| 827 | |
| 828 | let loopback = hostile("http://127.0.0.1:8080/private/token-shaped-path"); |
| 829 | assert_eq!(loopback.safe_endpoint_host_class(), "http loopback"); |
| 830 | } |
| 831 | |
| 832 | #[test] |
| 833 | fn wire_output_cap_is_read_only_from_the_finished_body() { |
| 834 | assert_eq!( |
| 835 | prepared(json!({"max_tokens": 1024})).wire_output_cap_tokens(), |
| 836 | Some(1024) |
| 837 | ); |
| 838 | assert_eq!( |
| 839 | prepared(json!({"max_completion_tokens": 2048})).wire_output_cap_tokens(), |
| 840 | Some(2048) |
| 841 | ); |
| 842 | assert_eq!( |
| 843 | prepared(json!({"model": "m"})).wire_output_cap_tokens(), |
| 844 | None |
| 845 | ); |
| 846 | } |
| 847 | |
| 848 | #[test] |
| 849 | fn reasoning_receipt_reads_the_finished_body_not_the_intent() { |
| 850 | // Kimi Code K3 strips `reasoning_effort` and writes nested thinking. |
| 851 | let kimi = prepared(json!({ |
| 852 | "model": "kimi-k3", |
| 853 | "messages": [], |
| 854 | "thinking": {"type": "enabled", "effort": "max"}, |
| 855 | })); |
| 856 | assert_eq!(kimi.reasoning.requested_effort.as_deref(), Some("high")); |
| 857 | assert_eq!(kimi.reasoning.wire_effort_string(), None); |
| 858 | assert_eq!( |
| 859 | kimi.reasoning.wire_controls, |
| 860 | vec![( |
| 861 | "thinking".to_string(), |
| 862 | json!({"type": "enabled", "effort": "max"}) |
| 863 | )] |
| 864 | ); |
| 865 | } |
| 866 | |
| 867 | #[test] |
| 868 | fn receipt_never_captures_message_or_prompt_fields() { |
| 869 | let leaky = prepared(json!({ |
| 870 | "model": "m", |
| 871 | "messages": [{"role": "user", "content": "SECRET PROMPT"}], |
| 872 | "instructions": "SECRET INSTRUCTIONS", |
| 873 | "reasoning_effort": "high", |
| 874 | })); |
| 875 | let rendered = format!("{:?}", leaky.reasoning); |
| 876 | assert!(!rendered.contains("SECRET PROMPT"), "{rendered}"); |
| 877 | assert!(!rendered.contains("SECRET INSTRUCTIONS"), "{rendered}"); |
| 878 | } |
| 879 | |
| 880 | #[test] |
| 881 | fn chat_view_folds_the_system_message_into_the_system_region() { |
| 882 | let request = prepared(json!({ |
| 883 | "model": "m", |
| 884 | "messages": [ |
| 885 | {"role": "system", "content": "SYS"}, |
| 886 | {"role": "user", "content": "hi"}, |
| 887 | {"role": "tool", "tool_call_id": "c1", "content": "OUT"}, |
| 888 | ], |
| 889 | "tools": [{"type": "function", "function": {"name": "a"}}], |
| 890 | "max_tokens": 100, |
| 891 | })); |
| 892 | let view = request.wire_view(); |
| 893 | assert!(view.system_bytes > 0); |
| 894 | assert_eq!(view.items.len(), 2, "system message is not a turn item"); |
| 895 | assert!(view.tool_result_bytes > 0); |
| 896 | assert_eq!(view.tool_count, 1); |
| 897 | assert!(view.framing_bytes > 0); |
| 898 | } |
| 899 | |
| 900 | #[test] |
| 901 | fn anthropic_and_responses_views_use_their_own_shapes() { |
| 902 | let anthropic_request = PreparedOutboundRequest::new( |
| 903 | WireDialect::AnthropicMessages, |
| 904 | endpoint(), |
| 905 | "claude".to_string(), |
| 906 | json!({ |
| 907 | "model": "claude", |
| 908 | "system": "SYS", |
| 909 | "messages": [ |
| 910 | {"role": "user", "content": [{"type": "tool_result", "content": "OUT"}]}, |
| 911 | {"role": "user", "content": [{"type": "image", "source": {"data": "AAA"}}]}, |
| 912 | ], |
| 913 | "tools": [{"name": "a"}, {"name": "b"}], |
| 914 | }), |
| 915 | None, |
| 916 | None, |
| 917 | CallerStreamMode::Streaming, |
| 918 | ); |
| 919 | let anthropic = anthropic_request.wire_view(); |
| 920 | assert!(anthropic.system_bytes > 0); |
| 921 | assert_eq!(anthropic.items.len(), 2); |
| 922 | assert!(anthropic.tool_result_bytes > 0); |
| 923 | assert_eq!(anthropic.attachment_count, 1); |
| 924 | assert_eq!(anthropic.tool_count, 2); |
| 925 | |
| 926 | let responses_request = PreparedOutboundRequest::new( |
| 927 | WireDialect::OpenAiResponses, |
| 928 | endpoint(), |
| 929 | "gpt".to_string(), |
| 930 | json!({ |
| 931 | "model": "gpt", |
| 932 | "instructions": "SYS", |
| 933 | "input": [ |
| 934 | {"type": "message", "role": "user", "content": [{"type": "input_text"}]}, |
| 935 | {"type": "function_call_output", "output": "OUT"}, |
| 936 | ], |
| 937 | "tools": [{"name": "a"}], |
| 938 | }), |
| 939 | None, |
| 940 | None, |
| 941 | CallerStreamMode::Streaming, |
| 942 | ); |
| 943 | let responses = responses_request.wire_view(); |
| 944 | assert!(responses.system_bytes > 0); |
| 945 | assert_eq!(responses.items.len(), 2); |
| 946 | assert!(responses.tool_result_bytes > 0); |
| 947 | assert_eq!(responses.tool_count, 1); |
| 948 | } |
| 949 | |
| 950 | /// The reviewed defect: nested reasoning shapes were invisible, so every |
| 951 | /// route that actually thinks hardest read as "no effort sent". |
| 952 | #[test] |
| 953 | fn nested_reasoning_efforts_are_read_from_every_dialect() { |
| 954 | let kimi = prepared(json!({ |
| 955 | "model": "kimi-k3", |
| 956 | "messages": [], |
| 957 | "thinking": {"type": "enabled", "effort": "max"}, |
| 958 | })); |
| 959 | assert_eq!( |
| 960 | kimi.reasoning.wire_effort(), |
| 961 | Some(("thinking.effort", "max")) |
| 962 | ); |
| 963 | assert!(kimi.reasoning.controls_reasoning()); |
| 964 | |
| 965 | let responses = PreparedOutboundRequest::new( |
| 966 | WireDialect::OpenAiResponses, |
| 967 | endpoint(), |
| 968 | "gpt".to_string(), |
| 969 | json!({ |
| 970 | "model": "gpt", |
| 971 | "input": [], |
| 972 | "reasoning": {"effort": "high", "summary": "auto"}, |
| 973 | "include": ["reasoning.encrypted_content"], |
| 974 | }), |
| 975 | None, |
| 976 | None, |
| 977 | CallerStreamMode::Streaming, |
| 978 | ); |
| 979 | assert_eq!( |
| 980 | responses.reasoning.wire_effort(), |
| 981 | Some(("reasoning.effort", "high")) |
| 982 | ); |
| 983 | assert!(responses.reasoning.controls_reasoning()); |
| 984 | |
| 985 | let anthropic = PreparedOutboundRequest::new( |
| 986 | WireDialect::AnthropicMessages, |
| 987 | endpoint(), |
| 988 | "claude".to_string(), |
| 989 | json!({ |
| 990 | "model": "claude", |
| 991 | "messages": [], |
| 992 | "output_config": {"effort": "low"}, |
| 993 | }), |
| 994 | None, |
| 995 | None, |
| 996 | CallerStreamMode::Streaming, |
| 997 | ); |
| 998 | assert_eq!( |
| 999 | anthropic.reasoning.wire_effort(), |
| 1000 | Some(("output_config.effort", "low")) |
| 1001 | ); |
| 1002 | |
| 1003 | // Flat still wins when the dialect uses it. |
| 1004 | let chat = prepared(json!({ |
| 1005 | "model": "m", |
| 1006 | "messages": [], |
| 1007 | "reasoning_effort": "medium", |
| 1008 | })); |
| 1009 | assert_eq!( |
| 1010 | chat.reasoning.wire_effort(), |
| 1011 | Some(("reasoning_effort", "medium")) |
| 1012 | ); |
| 1013 | } |
| 1014 | |
| 1015 | /// `include` discloses reasoning output; it does not request a tier. A |
| 1016 | /// body carrying only `include` must not read as a reasoning control. |
| 1017 | #[test] |
| 1018 | fn responses_include_alone_is_not_a_reasoning_control() { |
| 1019 | let disclosure_only = PreparedOutboundRequest::new( |
| 1020 | WireDialect::OpenAiResponses, |
| 1021 | endpoint(), |
| 1022 | "gpt".to_string(), |
| 1023 | json!({ |
| 1024 | "model": "gpt", |
| 1025 | "input": [], |
| 1026 | "include": ["reasoning.encrypted_content"], |
| 1027 | }), |
| 1028 | None, |
| 1029 | None, |
| 1030 | CallerStreamMode::Streaming, |
| 1031 | ); |
| 1032 | assert!( |
| 1033 | !disclosure_only.reasoning.wire_controls.is_empty(), |
| 1034 | "`include` is still disclosed on the receipt" |
| 1035 | ); |
| 1036 | assert!( |
| 1037 | !disclosure_only.reasoning.controls_reasoning(), |
| 1038 | "`include` alone must not read as a reasoning request" |
| 1039 | ); |
| 1040 | assert_eq!(disclosure_only.reasoning.wire_effort(), None); |
| 1041 | } |
| 1042 | |
| 1043 | fn assert_partition_exact(request: &PreparedOutboundRequest, what: &str) { |
| 1044 | let view = request.wire_view(); |
| 1045 | assert_eq!( |
| 1046 | view.body_bytes, |
| 1047 | request.canonical_body().len(), |
| 1048 | "{what}: the view must measure the bytes that would be POSTed" |
| 1049 | ); |
| 1050 | assert!( |
| 1051 | view.partition_is_exact(), |
| 1052 | "{what}: {} + {} + {} + {} != {}", |
| 1053 | view.system_bytes, |
| 1054 | view.tool_schema_bytes, |
| 1055 | view.item_bytes, |
| 1056 | view.framing_bytes, |
| 1057 | view.body_bytes |
| 1058 | ); |
| 1059 | assert!(view.tool_result_bytes <= view.item_bytes, "{what}"); |
| 1060 | assert!(view.attachment_bytes <= view.item_bytes, "{what}"); |
| 1061 | } |
| 1062 | |
| 1063 | /// The byte classes are published as exact facts, so they must account for |
| 1064 | /// every byte of the wire body — key names, brackets, and separators |
| 1065 | /// included — in every dialect and on both entry points. |
| 1066 | #[test] |
| 1067 | fn byte_classes_sum_to_the_whole_wire_body_in_every_dialect() { |
| 1068 | assert_partition_exact( |
| 1069 | &prepared(json!({ |
| 1070 | "model": "m", |
| 1071 | "messages": [ |
| 1072 | {"role": "system", "content": "SYS"}, |
| 1073 | {"role": "user", "content": "hi"}, |
| 1074 | {"role": "tool", "tool_call_id": "c1", "content": "OUT"}, |
| 1075 | ], |
| 1076 | "tools": [{"type": "function", "function": {"name": "a"}}], |
| 1077 | "tool_choice": {"type": "auto"}, |
| 1078 | "max_tokens": 100, |
| 1079 | "stream": true, |
| 1080 | })), |
| 1081 | "chat streaming", |
| 1082 | ); |
| 1083 | assert_partition_exact( |
| 1084 | &prepared(json!({ |
| 1085 | "model": "m", |
| 1086 | "messages": [{"role": "user", "content": "hi"}], |
| 1087 | "max_tokens": 100, |
| 1088 | })), |
| 1089 | "chat blocking (no tools, no system, no stream field)", |
| 1090 | ); |
| 1091 | assert_partition_exact( |
| 1092 | &prepared(json!({"model": "m", "messages": []})), |
| 1093 | "chat minimal", |
| 1094 | ); |
| 1095 | assert_partition_exact( |
| 1096 | &PreparedOutboundRequest::new( |
| 1097 | WireDialect::AnthropicMessages, |
| 1098 | endpoint(), |
| 1099 | "claude".to_string(), |
| 1100 | json!({ |
| 1101 | "model": "claude", |
| 1102 | "system": [{"type": "text", "text": "SYS"}], |
| 1103 | "messages": [ |
| 1104 | {"role": "user", "content": [{"type": "tool_result", "content": "OUT"}]}, |
| 1105 | {"role": "user", "content": [{"type": "image", "source": {"data": "AAA"}}]}, |
| 1106 | ], |
| 1107 | "tools": [{"name": "a"}], |
| 1108 | "stream": true, |
| 1109 | }), |
| 1110 | None, |
| 1111 | None, |
| 1112 | CallerStreamMode::Streaming, |
| 1113 | ), |
| 1114 | "anthropic streaming", |
| 1115 | ); |
| 1116 | assert_partition_exact( |
| 1117 | &PreparedOutboundRequest::new( |
| 1118 | WireDialect::OpenAiResponses, |
| 1119 | endpoint(), |
| 1120 | "gpt".to_string(), |
| 1121 | json!({ |
| 1122 | "model": "gpt", |
| 1123 | "instructions": "SYS", |
| 1124 | "input": [ |
| 1125 | {"type": "message", "role": "user", "content": [{"type": "input_text"}]}, |
| 1126 | {"type": "function_call_output", "output": "OUT"}, |
| 1127 | ], |
| 1128 | "tools": [{"name": "a"}], |
| 1129 | "reasoning": {"effort": "high"}, |
| 1130 | "stream": true, |
| 1131 | }), |
| 1132 | None, |
| 1133 | None, |
| 1134 | CallerStreamMode::Blocking, |
| 1135 | ), |
| 1136 | "responses blocking entry point (wire still streams)", |
| 1137 | ); |
| 1138 | } |
| 1139 | |
| 1140 | /// Mutating any region must keep the partition exact *and* move the class |
| 1141 | /// the mutation belongs to. A partition that stayed exact by dumping the |
| 1142 | /// difference into framing would be arithmetically true and useless. |
| 1143 | #[test] |
| 1144 | fn byte_classes_track_the_region_that_changed() { |
| 1145 | let base = json!({ |
| 1146 | "model": "m", |
| 1147 | "messages": [ |
| 1148 | {"role": "system", "content": "SYS"}, |
| 1149 | {"role": "user", "content": "hi"}, |
| 1150 | ], |
| 1151 | "tools": [{"type": "function", "function": {"name": "a"}}], |
| 1152 | "max_tokens": 100, |
| 1153 | }); |
| 1154 | let baseline = prepared(base.clone()); |
| 1155 | let baseline_view = baseline.wire_view(); |
| 1156 | |
| 1157 | let mut bigger_system = base.clone(); |
| 1158 | bigger_system["messages"][0]["content"] = json!("SYSTEM PROMPT, MUCH LONGER"); |
| 1159 | let request = prepared(bigger_system); |
| 1160 | let view = request.wire_view(); |
| 1161 | assert_partition_exact(&request, "grown system"); |
| 1162 | assert!(view.system_bytes > baseline_view.system_bytes); |
| 1163 | assert_eq!(view.item_bytes, baseline_view.item_bytes); |
| 1164 | |
| 1165 | let mut bigger_tools = base.clone(); |
| 1166 | bigger_tools["tools"][0]["function"]["parameters"] = json!({"type": "object"}); |
| 1167 | let request = prepared(bigger_tools); |
| 1168 | let view = request.wire_view(); |
| 1169 | assert_partition_exact(&request, "grown tool schema"); |
| 1170 | assert!(view.tool_schema_bytes > baseline_view.tool_schema_bytes); |
| 1171 | assert_ne!(view.tool_schema_sha256, baseline_view.tool_schema_sha256); |
| 1172 | |
| 1173 | let mut extra_message = base.clone(); |
| 1174 | extra_message["messages"] |
| 1175 | .as_array_mut() |
| 1176 | .expect("messages array") |
| 1177 | .push(json!({"role": "user", "content": "the hypothetical next prompt"})); |
| 1178 | let request = prepared(extra_message); |
| 1179 | let view = request.wire_view(); |
| 1180 | assert_partition_exact(&request, "appended message"); |
| 1181 | assert!(view.item_bytes > baseline_view.item_bytes); |
| 1182 | assert_eq!(view.system_bytes, baseline_view.system_bytes); |
| 1183 | |
| 1184 | let mut extra_framing = base; |
| 1185 | extra_framing["stream_options"] = json!({"include_usage": true}); |
| 1186 | let request = prepared(extra_framing); |
| 1187 | let view = request.wire_view(); |
| 1188 | assert_partition_exact(&request, "added framing field"); |
| 1189 | assert!(view.framing_bytes > baseline_view.framing_bytes); |
| 1190 | assert_eq!(view.item_bytes, baseline_view.item_bytes); |
| 1191 | } |
| 1192 | |
| 1193 | /// The prefix digest is derived from this hash, so a provider-side schema |
| 1194 | /// transform that leaves the logical catalog untouched must still move it. |
| 1195 | #[test] |
| 1196 | fn wire_tool_hash_tracks_dialect_schema_shaping() { |
| 1197 | let logical = prepared(json!({ |
| 1198 | "model": "m", |
| 1199 | "messages": [], |
| 1200 | "tools": [{"type": "function", "function": {"name": "a", "parameters": {"type": "object"}}}], |
| 1201 | })); |
| 1202 | let shaped = prepared(json!({ |
| 1203 | "model": "m", |
| 1204 | "messages": [], |
| 1205 | "tools": [{"type": "function", "function": { |
| 1206 | "name": "a", |
| 1207 | "parameters": {"type": "object", "additionalProperties": false}, |
| 1208 | "strict": true, |
| 1209 | }}], |
| 1210 | })); |
| 1211 | assert_ne!( |
| 1212 | logical.wire_view().tool_schema_sha256, |
| 1213 | shaped.wire_view().tool_schema_sha256, |
| 1214 | "strict-mode schema sanitizing must move the wire tool hash" |
| 1215 | ); |
| 1216 | |
| 1217 | let toolless = prepared(json!({"model": "m", "messages": []})); |
| 1218 | assert!(toolless.wire_view().tool_schema_sha256.is_empty()); |
| 1219 | } |
| 1220 | |
| 1221 | #[test] |
| 1222 | fn dialect_labels_are_stable() { |
| 1223 | assert_eq!( |
| 1224 | WireDialect::from_wire_format(WireFormat::ChatCompletions).as_str(), |
| 1225 | "chat-completions" |
| 1226 | ); |
| 1227 | assert_eq!( |
| 1228 | WireDialect::from_wire_format(WireFormat::AnthropicMessages).as_str(), |
| 1229 | "anthropic-messages" |
| 1230 | ); |
| 1231 | assert_eq!( |
| 1232 | WireDialect::from_wire_format(WireFormat::Responses).as_str(), |
| 1233 | "openai-responses" |
| 1234 | ); |
| 1235 | } |
| 1236 | } |
| 1237 | |
| 1238 | /// Per-dialect proof that `/preview-request` and production dispatch consume |
| 1239 | /// the same bytes. |
| 1240 | /// |
| 1241 | /// Each case builds a real client for a production route, prepares a request |
| 1242 | /// through [`CodewhaleClient::prepare_outbound_request`] — the value the |
| 1243 | /// transports send and the preview describes — and compares its whole-body |
| 1244 | /// hash against the dialect's own builder run over the identically |
| 1245 | /// pre-processed request. A divergence here means a second body builder has |
| 1246 | /// reappeared. |
| 1247 | #[cfg(test)] |
| 1248 | mod dialect_seam_tests { |
| 1249 | use super::*; |
| 1250 | use crate::config::{Config, ProviderConfig, ProvidersConfig}; |
| 1251 | use codewhale_models::Role; |
| 1252 | use codewhale_models::{ContentBlock, Message, MessageRequest, SystemPrompt, Tool}; |
| 1253 | use serde_json::json; |
| 1254 | |
| 1255 | use super::super::CodewhaleClient; |
| 1256 | |
| 1257 | fn tool(name: &str) -> Tool { |
| 1258 | Tool { |
| 1259 | tool_type: None, |
| 1260 | name: name.to_string(), |
| 1261 | description: format!("{name} description"), |
| 1262 | input_schema: json!({"type": "object", "properties": {}}), |
| 1263 | allowed_callers: None, |
| 1264 | defer_loading: None, |
| 1265 | input_examples: None, |
| 1266 | strict: None, |
| 1267 | cache_control: None, |
| 1268 | } |
| 1269 | } |
| 1270 | |
| 1271 | fn request(model: &str) -> MessageRequest { |
| 1272 | MessageRequest { |
| 1273 | model: model.to_string(), |
| 1274 | messages: vec![Message { |
| 1275 | role: Role::User, |
| 1276 | content: vec![ContentBlock::Text { |
| 1277 | text: "hello".to_string(), |
| 1278 | cache_control: None, |
| 1279 | }], |
| 1280 | }], |
| 1281 | max_tokens: 4096, |
| 1282 | system: Some(SystemPrompt::Text("BASE PROMPT".to_string())), |
| 1283 | tools: Some(vec![tool("read_file"), tool("Bash")]), |
| 1284 | tool_choice: Some(json!({"type": "auto"})), |
| 1285 | metadata: None, |
| 1286 | thinking: None, |
| 1287 | reasoning_effort: Some("high".to_string()), |
| 1288 | stream: Some(true), |
| 1289 | temperature: None, |
| 1290 | top_p: None, |
| 1291 | } |
| 1292 | } |
| 1293 | |
| 1294 | fn client(provider: &str, configure: impl FnOnce(&mut ProvidersConfig)) -> CodewhaleClient { |
| 1295 | let mut providers = ProvidersConfig::default(); |
| 1296 | configure(&mut providers); |
| 1297 | CodewhaleClient::new(&Config { |
| 1298 | provider: Some(provider.to_string()), |
| 1299 | providers: Some(providers), |
| 1300 | ..Config::default() |
| 1301 | }) |
| 1302 | .expect("client resolves for this route") |
| 1303 | } |
| 1304 | |
| 1305 | fn configured(api_key: &str, base_url: Option<&str>, model: &str) -> ProviderConfig { |
| 1306 | ProviderConfig { |
| 1307 | api_key: Some(api_key.to_string()), |
| 1308 | base_url: base_url.map(str::to_string), |
| 1309 | model: Some(model.to_string()), |
| 1310 | ..ProviderConfig::default() |
| 1311 | } |
| 1312 | } |
| 1313 | |
| 1314 | fn sha256(value: &str) -> String { |
| 1315 | crate::hashing::sha256_hex(value.as_bytes()) |
| 1316 | } |
| 1317 | |
| 1318 | /// The exact pre-processing `prepare_outbound_request` applies before the |
| 1319 | /// dialect builder runs. Reproduced here so the reference body is built |
| 1320 | /// from the same input, not from a differently-sanitized one. |
| 1321 | fn preprocessed(client: &CodewhaleClient, request: MessageRequest) -> MessageRequest { |
| 1322 | client |
| 1323 | .bind_request_to_protocol(client.prepare_model_bound_request(request)) |
| 1324 | .expect("protocol binding succeeds") |
| 1325 | .0 |
| 1326 | } |
| 1327 | |
| 1328 | #[test] |
| 1329 | fn output_cap_reaches_all_three_wire_dialects_with_reasoning_inside_allowance() { |
| 1330 | let _env = crate::test_support::lock_test_env(); |
| 1331 | for wire in ["chat-completions", "anthropic-messages", "responses"] { |
| 1332 | let config = Config { |
| 1333 | provider: Some("output-cap-fixture".into()), |
| 1334 | providers: Some(ProvidersConfig { |
| 1335 | custom: std::collections::HashMap::from([( |
| 1336 | "output-cap-fixture".into(), |
| 1337 | ProviderConfig { |
| 1338 | kind: Some("openai-compatible".into()), |
| 1339 | base_url: Some("http://127.0.0.1:18181/v1".into()), |
| 1340 | api_key: Some("fixture-output-cap".into()), |
| 1341 | model: Some("fixture-model".into()), |
| 1342 | wire: Some(wire.into()), |
| 1343 | ..Default::default() |
| 1344 | }, |
| 1345 | )]), |
| 1346 | ..Default::default() |
| 1347 | }), |
| 1348 | ..Config::default() |
| 1349 | }; |
| 1350 | let client = CodewhaleClient::new(&config).unwrap(); |
| 1351 | let mut request = request("fixture-model"); |
| 1352 | request.max_tokens = 1500; |
| 1353 | let prepared = client.prepare_outbound_request(request, true).unwrap(); |
| 1354 | assert_eq!(prepared.wire_output_cap_tokens(), Some(1500), "{wire}"); |
| 1355 | if let Some(thinking) = prepared |
| 1356 | .body |
| 1357 | .pointer("/thinking/budget_tokens") |
| 1358 | .and_then(Value::as_u64) |
| 1359 | { |
| 1360 | assert!( |
| 1361 | thinking < 1500, |
| 1362 | "reasoning must fit inside the shared allowance" |
| 1363 | ); |
| 1364 | } |
| 1365 | } |
| 1366 | } |
| 1367 | |
| 1368 | #[test] |
| 1369 | fn chat_completions_preview_matches_the_production_chat_builder() { |
| 1370 | let client = client("deepseek", |providers| { |
| 1371 | providers.deepseek = configured("sk-test-deepseek", None, "deepseek-chat"); |
| 1372 | }); |
| 1373 | let prepared = client |
| 1374 | .prepare_outbound_request(request("deepseek-chat"), true) |
| 1375 | .expect("chat request prepares"); |
| 1376 | assert_eq!(prepared.dialect, WireDialect::ChatCompletions); |
| 1377 | |
| 1378 | let reference = super::super::chat::build_chat_wire_body( |
| 1379 | &preprocessed(&client, request("deepseek-chat")), |
| 1380 | client.api_provider(), |
| 1381 | client.base_url(), |
| 1382 | true, |
| 1383 | None, |
| 1384 | ) |
| 1385 | .expect("reference body builds"); |
| 1386 | |
| 1387 | assert_eq!( |
| 1388 | prepared.body_sha256(), |
| 1389 | sha256(&canonical_json(&reference.body)) |
| 1390 | ); |
| 1391 | assert_eq!(prepared.wire_model, reference.model); |
| 1392 | assert!( |
| 1393 | prepared.body.get("tool_choice").is_none(), |
| 1394 | "DeepSeek thinking requests omit tool_choice on the final wire body" |
| 1395 | ); |
| 1396 | } |
| 1397 | |
| 1398 | #[test] |
| 1399 | fn kimi_code_keeps_its_own_shape_and_is_not_projected_through_plain_chat() { |
| 1400 | let client = client("moonshot", |providers| { |
| 1401 | providers.moonshot = configured( |
| 1402 | "sk-test-kimi", |
| 1403 | Some(crate::config::DEFAULT_KIMI_CODE_BASE_URL), |
| 1404 | crate::config::KIMI_CODE_K3_MODEL, |
| 1405 | ); |
| 1406 | }); |
| 1407 | let prepared = client |
| 1408 | .prepare_outbound_request(request(crate::config::KIMI_CODE_K3_MODEL), true) |
| 1409 | .expect("kimi code request prepares"); |
| 1410 | |
| 1411 | assert_eq!(prepared.dialect, WireDialect::ChatCompletions); |
| 1412 | assert_eq!(prepared.endpoint.shape, RouteShape::KimiCodeK3); |
| 1413 | // The route-specific shaper replaces flat `reasoning_effort` with the |
| 1414 | // nested `thinking.effort` dialect; the receipt must show that. |
| 1415 | assert_eq!(prepared.reasoning.wire_effort_string(), None); |
| 1416 | assert!( |
| 1417 | prepared |
| 1418 | .reasoning |
| 1419 | .wire_controls |
| 1420 | .iter() |
| 1421 | .any(|(key, _)| key == "thinking"), |
| 1422 | "{:?}", |
| 1423 | prepared.reasoning |
| 1424 | ); |
| 1425 | |
| 1426 | let reference = super::super::chat::build_chat_wire_body( |
| 1427 | &preprocessed(&client, request(crate::config::KIMI_CODE_K3_MODEL)), |
| 1428 | client.api_provider(), |
| 1429 | client.base_url(), |
| 1430 | true, |
| 1431 | None, |
| 1432 | ) |
| 1433 | .expect("reference body builds"); |
| 1434 | assert_eq!( |
| 1435 | prepared.body_sha256(), |
| 1436 | sha256(&canonical_json(&reference.body)) |
| 1437 | ); |
| 1438 | } |
| 1439 | |
| 1440 | /// Every production Anthropic Messages route: native Anthropic, the |
| 1441 | /// DeepSeek Messages route, and the MiniMax Messages route. Each shapes |
| 1442 | fn message(role: Role, text: &str) -> Message { |
| 1443 | Message { |
| 1444 | role, |
| 1445 | content: vec![ContentBlock::Text { |
| 1446 | text: text.to_string(), |
| 1447 | cache_control: None, |
| 1448 | }], |
| 1449 | } |
| 1450 | } |
| 1451 | |
| 1452 | /// Anthropic Messages has no in-transcript `system` role, but a compaction |
| 1453 | /// summary cannot be dropped or hoisted without changing transcript |
| 1454 | /// meaning. The seam keeps it in place and the adapter projects it onto a |
| 1455 | /// user message, one of the two roles the wire accepts. |
| 1456 | #[test] |
| 1457 | fn seam_preserves_an_in_transcript_system_message_on_anthropic() { |
| 1458 | let client = client("anthropic", |providers| { |
| 1459 | providers.anthropic = configured("sk-ant-test", None, "claude-sonnet-4-5"); |
| 1460 | }); |
| 1461 | let mut request = request("claude-sonnet-4-5"); |
| 1462 | request |
| 1463 | .messages |
| 1464 | .push(message(Role::System, "compaction summary")); |
| 1465 | |
| 1466 | let prepared = client |
| 1467 | .prepare_outbound_request(request, true) |
| 1468 | .expect("Anthropic projects positioned system history onto user"); |
| 1469 | let carried = prepared.body["messages"] |
| 1470 | .as_array() |
| 1471 | .expect("messages") |
| 1472 | .iter() |
| 1473 | .find(|message| { |
| 1474 | message["content"].as_array().is_some_and(|blocks| { |
| 1475 | blocks |
| 1476 | .iter() |
| 1477 | .any(|block| block["text"] == "compaction summary") |
| 1478 | }) |
| 1479 | }) |
| 1480 | .expect("compaction summary survives"); |
| 1481 | assert_eq!(carried["role"], "user"); |
| 1482 | } |
| 1483 | |
| 1484 | /// The dialects that have always dropped an unrepresentable role keep |
| 1485 | /// dropping it. Turning that into a hard failure would break live |
| 1486 | /// sessions; the point of the seam is to make the choice explicit, not to |
| 1487 | /// make every dialect strict. |
| 1488 | #[test] |
| 1489 | fn seam_lets_the_openai_shaped_dialects_keep_dropping_unknown_roles() { |
| 1490 | let client = client("deepseek", |providers| { |
| 1491 | providers.deepseek = configured("sk-test-deepseek", None, "deepseek-chat"); |
| 1492 | }); |
| 1493 | let mut request = request("deepseek-chat"); |
| 1494 | request.messages.push(message( |
| 1495 | Role::Unrecognized("future_role".to_string()), |
| 1496 | "from a newer build", |
| 1497 | )); |
| 1498 | |
| 1499 | let prepared = client |
| 1500 | .prepare_outbound_request(request, true) |
| 1501 | .expect("an unknown role must not fail a Chat Completions turn"); |
| 1502 | let body = serde_json::to_string(&prepared.body).expect("serialize body"); |
| 1503 | assert!( |
| 1504 | !body.contains("from a newer build"), |
| 1505 | "an unknown role must not reach the wire: {body}" |
| 1506 | ); |
| 1507 | assert!(!body.contains("\"future_role\""), "{body}"); |
| 1508 | } |
| 1509 | |
| 1510 | /// thinking differently, so each is checked against its own builder run. |
| 1511 | #[test] |
| 1512 | fn anthropic_messages_preview_matches_the_production_messages_builder() { |
| 1513 | type ProviderCase = ( |
| 1514 | &'static str, |
| 1515 | &'static str, |
| 1516 | Box<dyn Fn(&mut ProvidersConfig)>, |
| 1517 | ); |
| 1518 | |
| 1519 | let cases: Vec<ProviderCase> = vec![ |
| 1520 | ( |
| 1521 | "anthropic", |
| 1522 | "claude-sonnet-4-5", |
| 1523 | Box::new(|providers: &mut ProvidersConfig| { |
| 1524 | providers.anthropic = configured("sk-ant-test", None, "claude-sonnet-4-5"); |
| 1525 | }), |
| 1526 | ), |
| 1527 | ( |
| 1528 | "deepseek-anthropic", |
| 1529 | "deepseek-v4", |
| 1530 | Box::new(|providers: &mut ProvidersConfig| { |
| 1531 | providers.deepseek_anthropic = |
| 1532 | configured("sk-test-deepseek-anthropic", None, "deepseek-v4"); |
| 1533 | }), |
| 1534 | ), |
| 1535 | ( |
| 1536 | "minimax-anthropic", |
| 1537 | "MiniMax-M3", |
| 1538 | Box::new(|providers: &mut ProvidersConfig| { |
| 1539 | providers.minimax_anthropic = |
| 1540 | configured("sk-test-minimax-anthropic", None, "MiniMax-M3"); |
| 1541 | }), |
| 1542 | ), |
| 1543 | ]; |
| 1544 | |
| 1545 | for (provider, model, configure) in cases { |
| 1546 | let client = client(provider, |providers| configure(providers)); |
| 1547 | let prepared = client |
| 1548 | .prepare_outbound_request(request(model), true) |
| 1549 | .unwrap_or_else(|error| panic!("{provider} request prepares: {error}")); |
| 1550 | |
| 1551 | assert_eq!( |
| 1552 | prepared.dialect, |
| 1553 | WireDialect::AnthropicMessages, |
| 1554 | "{provider} must keep the Messages dialect, not be projected through Chat" |
| 1555 | ); |
| 1556 | |
| 1557 | let reference = |
| 1558 | client.build_anthropic_body(&preprocessed(&client, request(model)), true); |
| 1559 | assert_eq!( |
| 1560 | prepared.body_sha256(), |
| 1561 | sha256(&canonical_json(&reference)), |
| 1562 | "{provider} preview body must hash identically to the production builder" |
| 1563 | ); |
| 1564 | assert_eq!( |
| 1565 | prepared |
| 1566 | .body |
| 1567 | .get("tool_choice") |
| 1568 | .and_then(|value| value.get("type")) |
| 1569 | .and_then(serde_json::Value::as_str), |
| 1570 | Some("auto"), |
| 1571 | "{provider} tool_choice must come from the final Messages body" |
| 1572 | ); |
| 1573 | |
| 1574 | // The Messages dialect never carries a flat `reasoning_effort`; |
| 1575 | // the receipt must reflect the dialect's own controls. |
| 1576 | assert_eq!(prepared.reasoning.wire_effort_string(), None, "{provider}"); |
| 1577 | assert_eq!( |
| 1578 | prepared.reasoning.requested_effort.as_deref(), |
| 1579 | Some("high"), |
| 1580 | "{provider}" |
| 1581 | ); |
| 1582 | } |
| 1583 | } |
| 1584 | |
| 1585 | /// The official plan route requires Codewhale's own protected grant. |
| 1586 | fn codex_client() -> CodewhaleClient { |
| 1587 | let _env_lock = crate::test_support::lock_test_env(); |
| 1588 | let home = tempfile::tempdir().expect("isolated ChatGPT credential home"); |
| 1589 | let root = home |
| 1590 | .path() |
| 1591 | .canonicalize() |
| 1592 | .expect("canonical credential home"); |
| 1593 | let _home = crate::test_support::EnvVarGuard::set("CODEWHALE_HOME", &root); |
| 1594 | let mut config = Config { |
| 1595 | provider: Some("openai-codex".to_string()), |
| 1596 | ..Config::default() |
| 1597 | }; |
| 1598 | config |
| 1599 | .provider_config_for_mut(&config.test_identity_for_kind(ProviderKind::OpenaiCodex)) |
| 1600 | .unwrap() |
| 1601 | .model = Some("gpt-5-codex".into()); |
| 1602 | crate::oauth::install_test_chatgpt_registration(&mut config).expect("official test grant"); |
| 1603 | CodewhaleClient::new(&config).expect("official ChatGPT client") |
| 1604 | } |
| 1605 | |
| 1606 | #[test] |
| 1607 | fn responses_preview_matches_the_production_responses_builder() { |
| 1608 | let client = codex_client(); |
| 1609 | let prepared = client |
| 1610 | .prepare_outbound_request(request("gpt-5-codex"), true) |
| 1611 | .expect("responses request prepares"); |
| 1612 | |
| 1613 | assert_eq!(prepared.dialect, WireDialect::OpenAiResponses); |
| 1614 | assert_eq!(prepared.endpoint.shape, RouteShape::CodexResponses); |
| 1615 | assert_eq!(prepared.endpoint.url, "https://api.openai.com/v1/responses"); |
| 1616 | assert_eq!(prepared.body["tools"][0]["type"], "namespace"); |
| 1617 | assert_eq!(prepared.body["tools"][0]["name"], "codewhale"); |
| 1618 | assert_eq!( |
| 1619 | prepared.body["tools"][0]["tools"].as_array().unwrap().len(), |
| 1620 | 2 |
| 1621 | ); |
| 1622 | assert_eq!( |
| 1623 | prepared.wire_view().tool_count, |
| 1624 | 2, |
| 1625 | "namespace wrappers are not executable tools" |
| 1626 | ); |
| 1627 | |
| 1628 | let reference = super::super::responses::build_responses_body_for_provider( |
| 1629 | &preprocessed(&client, request("gpt-5-codex")), |
| 1630 | ProviderKind::OpenaiCodex, |
| 1631 | client.chatgpt_reasoning_api.as_deref(), |
| 1632 | ); |
| 1633 | assert_eq!(prepared.body_sha256(), sha256(&canonical_json(&reference))); |
| 1634 | assert_eq!(prepared.body.get("tool_choice"), Some(&json!("auto"))); |
| 1635 | } |
| 1636 | |
| 1637 | #[test] |
| 1638 | fn responses_prepared_request_preserves_full_history_for_its_grant() { |
| 1639 | let client = codex_client(); |
| 1640 | let scope = client.chatgpt_reasoning_api.as_ref().unwrap(); |
| 1641 | let mut request = request("gpt-5-codex"); |
| 1642 | for id in ["first", "second"] { |
| 1643 | request.messages.push(codewhale_models::Message { |
| 1644 | role: codewhale_models::Role::Assistant, |
| 1645 | content: vec![codewhale_models::ContentBlock::Thinking { |
| 1646 | thinking: "readable-reasoning-must-stay-local".into(), |
| 1647 | signature: None, |
| 1648 | state: Some(codewhale_models::OpaqueReasoningState { |
| 1649 | provider: "openai-codex".into(), |
| 1650 | api: scope.clone(), |
| 1651 | model: "gpt-5-codex".into(), |
| 1652 | id: Some(id.into()), |
| 1653 | encrypted_content: format!("opaque-{id}"), |
| 1654 | }), |
| 1655 | }], |
| 1656 | }); |
| 1657 | } |
| 1658 | let prepared = client |
| 1659 | .prepare_outbound_request(request.clone(), true) |
| 1660 | .unwrap(); |
| 1661 | let reasoning: Vec<_> = prepared.body["input"] |
| 1662 | .as_array() |
| 1663 | .unwrap() |
| 1664 | .iter() |
| 1665 | .filter(|item| item["type"] == "reasoning") |
| 1666 | .collect(); |
| 1667 | assert_eq!(reasoning.len(), 2); |
| 1668 | assert_eq!(reasoning[0]["encrypted_content"], "opaque-first"); |
| 1669 | assert_eq!(reasoning[1]["encrypted_content"], "opaque-second"); |
| 1670 | assert!( |
| 1671 | !prepared |
| 1672 | .body |
| 1673 | .to_string() |
| 1674 | .contains("readable-reasoning-must-stay-local") |
| 1675 | ); |
| 1676 | let mut different_grant = client.clone(); |
| 1677 | different_grant.chatgpt_reasoning_api = Some("openai-responses-siwc-v1:other".into()); |
| 1678 | let rejected = different_grant |
| 1679 | .prepare_outbound_request(request, true) |
| 1680 | .unwrap(); |
| 1681 | assert!(!rejected.body.to_string().contains("opaque-first")); |
| 1682 | assert!(!rejected.body.to_string().contains("opaque-second")); |
| 1683 | } |
| 1684 | |
| 1685 | #[test] |
| 1686 | fn every_dialect_reports_a_distinct_body_hash_for_the_same_logical_request() { |
| 1687 | // Guards against the reviewed failure mode: projecting every route |
| 1688 | // through the Chat builder would make these collide. |
| 1689 | let chat = client("deepseek", |providers| { |
| 1690 | providers.deepseek = configured("sk-test-deepseek", None, "deepseek-chat"); |
| 1691 | }) |
| 1692 | .prepare_outbound_request(request("deepseek-chat"), true) |
| 1693 | .expect("chat prepares"); |
| 1694 | let codex = codex_client(); |
| 1695 | let responses = codex |
| 1696 | .prepare_outbound_request(request("gpt-5-codex"), true) |
| 1697 | .expect("responses prepares"); |
| 1698 | |
| 1699 | assert_ne!(chat.dialect, responses.dialect); |
| 1700 | assert_ne!(chat.body_sha256(), responses.body_sha256()); |
| 1701 | } |
| 1702 | |
| 1703 | #[test] |
| 1704 | fn streaming_and_blocking_bodies_are_distinguished_not_conflated() { |
| 1705 | let client = client("deepseek", |providers| { |
| 1706 | providers.deepseek = configured("sk-test-deepseek", None, "deepseek-chat"); |
| 1707 | }); |
| 1708 | let streaming = client |
| 1709 | .prepare_outbound_request(request("deepseek-chat"), true) |
| 1710 | .expect("streaming prepares"); |
| 1711 | let blocking = client |
| 1712 | .prepare_outbound_request(request("deepseek-chat"), false) |
| 1713 | .expect("blocking prepares"); |
| 1714 | |
| 1715 | assert_eq!(streaming.entrypoint, CallerStreamMode::Streaming); |
| 1716 | assert_eq!(blocking.entrypoint, CallerStreamMode::Blocking); |
| 1717 | // Chat is the dialect where caller mode and wire fact agree: the |
| 1718 | // streaming body sets `stream: true`, the blocking body omits it. |
| 1719 | assert_eq!(streaming.wire_stream_field(), Some(true)); |
| 1720 | assert_eq!(blocking.wire_stream_field(), None); |
| 1721 | assert_ne!(streaming.body_sha256(), blocking.body_sha256()); |
| 1722 | } |
| 1723 | |
| 1724 | /// #1004 review finding: the Responses blocking entry point opens an SSE |
| 1725 | /// stream and folds it into one response, so its body says |
| 1726 | /// `"stream": true` while the caller mode is blocking. The manifest must |
| 1727 | /// read the body, never the caller mode. |
| 1728 | #[test] |
| 1729 | fn responses_wire_streaming_is_read_from_the_body_not_the_caller_mode() { |
| 1730 | let client = codex_client(); |
| 1731 | let blocking = client |
| 1732 | .prepare_outbound_request(request("gpt-5-codex"), false) |
| 1733 | .expect("blocking responses prepares"); |
| 1734 | |
| 1735 | assert_eq!(blocking.entrypoint, CallerStreamMode::Blocking); |
| 1736 | assert_eq!( |
| 1737 | blocking.wire_stream_field(), |
| 1738 | Some(true), |
| 1739 | "the Responses blocking path genuinely sends an SSE body" |
| 1740 | ); |
| 1741 | |
| 1742 | let streaming = client |
| 1743 | .prepare_outbound_request(request("gpt-5-codex"), true) |
| 1744 | .expect("streaming responses prepares"); |
| 1745 | assert_eq!(streaming.wire_stream_field(), Some(true)); |
| 1746 | assert_eq!( |
| 1747 | streaming.body_sha256(), |
| 1748 | blocking.body_sha256(), |
| 1749 | "the two Responses entry points send the same bytes; only the \ |
| 1750 | caller mode differs" |
| 1751 | ); |
| 1752 | } |
| 1753 | |
| 1754 | #[test] |
| 1755 | fn preparation_is_deterministic_across_repeated_calls() { |
| 1756 | let client = client("deepseek", |providers| { |
| 1757 | providers.deepseek = configured("sk-test-deepseek", None, "deepseek-chat"); |
| 1758 | }); |
| 1759 | let first = client |
| 1760 | .prepare_outbound_request(request("deepseek-chat"), true) |
| 1761 | .expect("first prepares"); |
| 1762 | let second = client |
| 1763 | .prepare_outbound_request(request("deepseek-chat"), true) |
| 1764 | .expect("second prepares"); |
| 1765 | assert_eq!(first.body_sha256(), second.body_sha256()); |
| 1766 | assert_eq!(first.endpoint_fingerprint(), second.endpoint_fingerprint()); |
| 1767 | } |
| 1768 | } |
| 1769 |