返回 CodeWhale
lib.rs
根目录 / crates / models / src / lib.rs
1 //! API request/response models for `DeepSeek` and OpenAI-compatible endpoints.
2
3 use serde::{Deserialize, Serialize};
4
5 /// Context window used only for legacy DeepSeek model IDs that do not name a
6 /// newer V4 alias and do not carry an explicit `*k` suffix.
7 pub const LEGACY_DEEPSEEK_CONTEXT_WINDOW_TOKENS: u32 = 128_000;
8 pub use codewhale_config::catalog::reviewed::constants::DEEPSEEK_V4_CONTEXT_WINDOW_TOKENS;
9 /// Conservative Kimi Code K3 context baseline. The membership route's real
10 /// context is plan-tier dependent (verified 2026-07-20 from
11 /// <https://www.kimi.com/code/docs/en/kimi-code/models>): Moderato gets 256K,
12 /// while Allegretto and above get up to 1M. Bare `k3` therefore keeps this
13 /// safe floor everywhere; higher plan entitlements must come from an explicit
14 /// provider `context_window` configuration or fresh provider facts while
15 /// preserving the `k3` wire id.
16 pub const KIMI_CODE_K3_CONTEXT_WINDOW_TOKENS: u32 = 262_144;
17 /// Kimi K3 context window on the open platform (`kimi-k3` pay-as-you-go).
18 /// Verified 2026-07-20 from <https://platform.kimi.ai/docs/guide/kimi-k3-quickstart>
19 /// (1,048,576 tokens). Max output is a separate fact below and must never be
20 /// conflated with this window.
21 pub use codewhale_config::catalog::reviewed::constants::KIMI_K3_CONTEXT_WINDOW_TOKENS;
22 /// Conservative K3 default generation ceiling. The direct Kimi API defaults
23 /// `max_completion_tokens` to 131,072, while its documented route maximum is
24 /// a separate exact-route fact below. Membership and neighboring routes do
25 /// not inherit that direct-platform maximum.
26 pub use codewhale_config::catalog::reviewed::constants::KIMI_K3_DEFAULT_MAX_COMPLETION_TOKENS;
27 /// Documented maximum output for the exact direct Kimi K3 API route.
28 ///
29 /// Source: <https://platform.kimi.ai/docs/guide/kimi-k3-quickstart> (verified 2026-07-20).
30 pub const DIRECT_KIMI_K3_MAX_OUTPUT_TOKENS: u32 = 1_048_576;
31 /// Last-resort compaction trigger when [`context_window_for_model`] returns
32 /// `None` (an unrecognised model id). v0.8.11 raised this from `50_000` to
33 /// `102_400` (80% of [`LEGACY_DEEPSEEK_CONTEXT_WINDOW_TOKENS`]) so unknown
34 /// models inherit the same late-trigger discipline as V4 instead of paying
35 /// the prefix-cache hit at 5% of the V4 window. Known DeepSeek / Claude
36 /// models resolve to their own scaled value via
37 /// `compaction_threshold_for_model` (#664).
38 pub const DEFAULT_COMPACTION_TOKEN_THRESHOLD: usize = 102_400;
39 pub fn canonical_official_deepseek_model_id(model: &str) -> Option<&'static str> {
40 codewhale_config::catalog::reviewed::official_deepseek_model_id(model)
41 }
42
43 #[cfg(any(test, feature = "test-support"))]
44 const COMPACTION_THRESHOLD_PERCENT: u32 = 80;
45
46 // === Core Message Types ===
47
48 // Keep the historical TUI path stable while the production request DTOs are
49 // owned by `codewhale-core`. Existing transports and response decoders do not
50 // need a flag day, and headless callers can depend on core directly.
51 // Some process-test crates include this module privately and exercise only a
52 // subset of the compatibility surface, so their crate-local dead-import view
53 // is not evidence that a re-export can be removed.
54 #[allow(unused_imports)]
55 pub use codewhale_core::request::{
56 CacheControl, ContentBlock, INTERRUPTED_ASSISTANT_CONTEXT_PREFIX, INTERRUPTED_ASSISTANT_ROLE,
57 ImageUrlContent, Message, MessageRequest, OpaqueReasoningState, SystemBlock, SystemPrompt,
58 Tool, ToolCallKey, ToolCaller,
59 };
60 #[allow(unused_imports)]
61 pub use codewhale_core::role::Role;
62
63 /// Container metadata for code-execution style server tools.
64 #[derive(Debug, Serialize, Deserialize, Clone)]
65 pub struct ContainerInfo {
66 pub id: String,
67 #[serde(skip_serializing_if = "Option::is_none")]
68 pub expires_at: Option<String>,
69 }
70
71 /// Server-side tool usage counters.
72 #[derive(Debug, Serialize, Deserialize, Clone, Default, PartialEq, Eq)]
73 pub struct ServerToolUsage {
74 #[serde(skip_serializing_if = "Option::is_none")]
75 pub code_execution_requests: Option<u32>,
76 #[serde(skip_serializing_if = "Option::is_none")]
77 pub tool_search_requests: Option<u32>,
78 }
79
80 /// Response payload for a message request.
81 #[derive(Debug, Serialize, Deserialize, Clone)]
82 pub struct MessageResponse {
83 pub id: String,
84 pub r#type: String,
85 pub role: String,
86 pub content: Vec<ContentBlock>,
87 pub model: String,
88 pub stop_reason: Option<String>,
89 pub stop_sequence: Option<String>,
90 #[serde(skip_serializing_if = "Option::is_none")]
91 pub container: Option<ContainerInfo>,
92 pub usage: Usage,
93 }
94
95 /// True when the provider ended generation because its output allowance was
96 /// exhausted. Providers use several wire spellings for the same condition.
97 #[must_use]
98 pub fn is_output_limit_stop_reason(reason: Option<&str>) -> bool {
99 reason.is_some_and(|reason| {
100 let reason = reason
101 .trim()
102 .strip_prefix("incomplete:")
103 .unwrap_or_else(|| reason.trim());
104 matches!(
105 reason.to_ascii_lowercase().as_str(),
106 "length" | "max_tokens" | "max_output_tokens"
107 )
108 })
109 }
110
111 /// True when the provider explicitly reported that it did not complete the
112 /// response. Responses API reasons carry an `incomplete:` prefix so unknown
113 /// future reasons cannot accidentally be accepted as a finished answer.
114 #[must_use]
115 pub fn is_incomplete_stop_reason(reason: Option<&str>) -> bool {
116 is_output_limit_stop_reason(reason)
117 || reason.is_some_and(|reason| {
118 let reason = reason.trim().to_ascii_lowercase();
119 reason.starts_with("incomplete:")
120 || matches!(
121 reason.as_str(),
122 "content_filter" | "model_context_window_exceeded"
123 )
124 })
125 }
126
127 #[must_use]
128 pub fn stop_reason_detail(reason: Option<&str>) -> &str {
129 reason
130 .map(str::trim)
131 .and_then(|reason| reason.strip_prefix("incomplete:").or(Some(reason)))
132 .filter(|reason| !reason.is_empty())
133 .unwrap_or("unknown")
134 }
135
136 /// Token usage metadata for a response.
137 #[derive(Debug, Serialize, Deserialize, Clone, Default, PartialEq, Eq)]
138 pub struct Usage {
139 pub input_tokens: u32,
140 pub output_tokens: u32,
141 #[serde(skip_serializing_if = "Option::is_none")]
142 pub prompt_cache_hit_tokens: Option<u32>,
143 #[serde(skip_serializing_if = "Option::is_none")]
144 pub prompt_cache_miss_tokens: Option<u32>,
145 /// Cache-creation / cache-write tokens (Anthropic `cache_creation_input_tokens`).
146 /// Billed at the cache-write rate when the pricing row publishes one (#4318).
147 #[serde(skip_serializing_if = "Option::is_none")]
148 pub prompt_cache_write_tokens: Option<u32>,
149 #[serde(skip_serializing_if = "Option::is_none")]
150 pub reasoning_tokens: Option<u32>,
151 /// Approximate input tokens spent re-sending prior `reasoning_content`
152 /// across user-message boundaries in DeepSeek V4 thinking-mode tool-calling
153 /// turns (V4 §5.1.1 "Interleaved Thinking"). Estimated client-side at
154 /// ~4 chars/token from the outgoing request body, before the model sees it.
155 #[serde(skip_serializing_if = "Option::is_none")]
156 pub reasoning_replay_tokens: Option<u32>,
157 #[serde(skip_serializing_if = "Option::is_none")]
158 pub server_tool_use: Option<ServerToolUsage>,
159 }
160
161 /// Map known models to their approximate context window sizes.
162 ///
163 /// Exact catalog and recognized model facts take precedence. Otherwise an
164 /// explicit `_Nk` suffix supplies an unverified hint for self-hosted models.
165 /// Unrecognized DeepSeek family names remain unknown.
166 #[must_use]
167 pub fn context_window_for_model(model: &str) -> Option<u32> {
168 codewhale_config::catalog::reviewed::intrinsic_model(model)
169 .and_then(|row| row.context_window)
170 .or_else(|| {
171 reviewed_snapshot_model(model).and_then(|id| {
172 codewhale_config::catalog::reviewed::intrinsic_model(id)
173 .and_then(|row| row.context_window)
174 })
175 })
176 .or_else(|| explicit_context_window_hint(&model.to_ascii_lowercase()))
177 }
178
179 #[must_use]
180 pub fn max_output_tokens_for_model(model: &str) -> Option<u32> {
181 codewhale_config::catalog::reviewed::intrinsic_model(model)
182 .and_then(|row| row.generation_default.or(row.max_output))
183 .or_else(|| {
184 reviewed_snapshot_model(model).and_then(|id| {
185 codewhale_config::catalog::reviewed::intrinsic_model(id)
186 .and_then(|row| row.generation_default.or(row.max_output))
187 })
188 })
189 }
190
191 /// Catalog-first reasoning capability. `None` means no catalog row and no
192 /// remaining cited fallback — unknown, not "not a reasoning model".
193 ///
194 /// Prefer this over [`model_supports_reasoning`] when the caller can surface
195 /// unknown the way unknown cost already is. The bool wrapper still defaults
196 /// unknown to `false` for existing stream/UI gates.
197 #[must_use]
198 pub fn model_reasoning_capability(model: &str) -> Option<bool> {
199 codewhale_config::catalog::reviewed::intrinsic_model(model)
200 .and_then(|row| row.reasoning)
201 .or_else(|| {
202 reviewed_snapshot_model(model).and_then(|id| {
203 codewhale_config::catalog::reviewed::intrinsic_model(id)
204 .and_then(|row| row.reasoning)
205 })
206 })
207 }
208
209 #[must_use]
210 pub fn model_supports_reasoning(model: &str) -> bool {
211 model_reasoning_capability(model).unwrap_or(false)
212 }
213
214 /// Contributor tier of Muse Spark 1.2 is a distinct selectable id with
215 /// its own wire model (`muse-spark-1.2-contributor`) and cheaper billing in
216 /// exchange for training-data opt-in. Do not collapse it to the standard tier.
217 #[must_use]
218 pub fn effective_muse_wire_id(model: &str) -> &str {
219 model
220 }
221
222 #[must_use]
223 pub fn model_is_openai_reasoning_family(model: &str) -> bool {
224 let lower = model.to_ascii_lowercase();
225 let reviewed = codewhale_config::catalog::reviewed::bundled_reviewed();
226 reviewed.openai_reasoning_ids.contains_key(&lower)
227 // Snapshot resolution spans every reviewed family; the resolved
228 // target must itself be an OpenAI reasoning row, or a DeepSeek
229 // snapshot would be relabelled as one.
230 || reviewed_snapshot_model(&lower)
231 .is_some_and(|target| reviewed.openai_reasoning_ids.contains_key(target))
232 }
233
234 pub fn is_openai_gpt_56_api_model(model_lower: &str) -> bool {
235 codewhale_config::catalog::reviewed::bundled_reviewed()
236 .openai_reasoning_ids
237 .get(model_lower)
238 .is_some_and(|kind| kind == "gpt_56")
239 }
240
241 pub fn has_date_snapshot_suffix(model_lower: &str, prefix: &str) -> bool {
242 let Some(rest) = model_lower.strip_prefix(prefix) else {
243 return false;
244 };
245 let bytes = rest.as_bytes();
246 bytes.len() == 10
247 && bytes[4] == b'-'
248 && bytes[7] == b'-'
249 && bytes
250 .iter()
251 .enumerate()
252 .all(|(idx, byte)| idx == 4 || idx == 7 || byte.is_ascii_digit())
253 }
254
255 /// Resolve a snapshot or variant model id to the reviewed intrinsic row it
256 /// denotes, without inventing facts for names the catalog does not own.
257 ///
258 /// Two documented contracts, applied in order:
259 /// - the authored `snapshot_prefixes` rows (`gpt-5.5-<YYYY-MM-DD>`);
260 /// - a trailing date stamp (`-YYYY-MM-DD` or `-MMDD`) or an explicit variant
261 /// marker (`-vision-exp`, `-vision`, `-exp`), accepted only when stripping
262 /// it leaves the exact id of a reviewed intrinsic row
263 /// (`deepseek-v4-pro-0813` denotes `deepseek-v4-pro`;
264 /// `deepseek-v4-flash-vision` denotes `deepseek-v4-flash`).
265 /// The compact `-YYYYMMDD` shape is deliberately not inferred: the
266 /// sibling-metadata contract rejects it.
267 ///
268 /// Every layer is resolved against the bundled reviewed catalog only. An
269 /// unrecognized name still returns `None`: the unknown stays observable.
270 fn reviewed_snapshot_model(model: &str) -> Option<&'static str> {
271 let reviewed = codewhale_config::catalog::reviewed::bundled_reviewed();
272 let via_authored_contract = |candidate: &str| {
273 reviewed
274 .snapshot_prefixes
275 .iter()
276 .find_map(|(prefix, target)| {
277 has_date_snapshot_suffix(candidate, prefix).then_some(target.as_str())
278 })
279 };
280 let mut candidate = model.to_ascii_lowercase();
281 for _ in 0..3 {
282 if let Some(target) = via_authored_contract(&candidate) {
283 return Some(target);
284 }
285 let stripped = strip_snapshot_or_variant_suffix(&candidate)?;
286 if let Some((key, _)) = reviewed.intrinsic.get_key_value(&stripped) {
287 return Some(key.as_str());
288 }
289 candidate = stripped;
290 }
291 None
292 }
293
294 /// One snapshot/variant normalization layer, longest marker first. The
295 /// caller re-resolves the shortened id against the reviewed catalog and
296 /// never accepts a shortening on its own.
297 fn strip_snapshot_or_variant_suffix(id: &str) -> Option<String> {
298 for marker in ["-vision-exp", "-vision", "-exp"] {
299 if let Some(base) = id.strip_suffix(marker)
300 && !base.is_empty()
301 {
302 return Some(base.to_string());
303 }
304 }
305 strip_date_stamp(id)
306 }
307
308 /// Strip one trailing date stamp: `-YYYY-MM-DD` or `-MMDD` (the reviewed
309 /// snapshot conventions). The compact `-YYYYMMDD` form is deliberately not
310 /// inferred: the sibling-metadata contract rejects that shape (see
311 /// `unrecognized_deepseek_models_do_not_inherit_sibling_metadata`).
312 fn strip_date_stamp(id: &str) -> Option<String> {
313 // `-YYYY-MM-DD`: a ten-character tail preceded by its own dash. The split
314 // index is a byte offset, so it must land on a char boundary: a multi-byte
315 // model id would panic in `split_at` otherwise.
316 if id.len() > 11 && id.is_char_boundary(id.len() - 10) {
317 let (head, tail) = id.split_at(id.len() - 10);
318 if let Some(head) = head.strip_suffix('-')
319 && !head.is_empty()
320 && has_date_snapshot_suffix(tail, "")
321 {
322 return Some(head.to_string());
323 }
324 }
325 let (head, tail) = id.rsplit_once('-')?;
326 if head.is_empty() || tail.is_empty() {
327 return None;
328 }
329 let digits = tail.as_bytes();
330 if digits.len() == 4
331 && digits.iter().all(u8::is_ascii_digit)
332 && valid_month_day(&digits[0..2], &digits[2..4])
333 {
334 return Some(head.to_string());
335 }
336 None
337 }
338
339 /// Validate a two-digit month/day pair (ranges only, no calendar math).
340 fn valid_month_day(month_digits: &[u8], day_digits: &[u8]) -> bool {
341 let two = |bytes: &[u8]| -> Option<u32> {
342 (bytes.len() == 2 && bytes.iter().all(u8::is_ascii_digit))
343 .then(|| u32::from(bytes[0] - b'0') * 10 + u32::from(bytes[1] - b'0'))
344 };
345 match (two(month_digits), two(day_digits)) {
346 (Some(month), Some(day)) => (1..=12).contains(&month) && (1..=31).contains(&day),
347 _ => false,
348 }
349 }
350
351 /// The context window a model name's `_Nk` suffix advertises, when the
352 /// catalog does not already describe the model (#5441).
353 ///
354 /// Exposed separately from `explicit_context_window_hint` because the
355 /// honesty surfaces need to know *whether the number they are holding came
356 /// from the name* — a naming convention the serving engine may ignore is not
357 /// a fact about the route, and every surface that shows such a window must
358 /// mark it unverified.
359 #[must_use]
360 pub fn name_suffix_context_window_hint(model: &str) -> Option<u32> {
361 if codewhale_config::catalog::reviewed::intrinsic_model(model)
362 .and_then(|row| row.context_window)
363 .is_some()
364 || reviewed_snapshot_model(model).is_some()
365 {
366 return None;
367 }
368 explicit_context_window_hint(&model.to_ascii_lowercase())
369 }
370
371 /// Parse an explicit `_Nk` context-window hint from a model name (vendor
372 /// agnostic). Returns the window in tokens for `N` in `8..=1024`.
373 fn explicit_context_window_hint(model_lower: &str) -> Option<u32> {
374 let bytes = model_lower.as_bytes();
375 let mut i = 0usize;
376 while i < bytes.len() {
377 if bytes[i].is_ascii_digit() {
378 let start = i;
379 while i < bytes.len() && bytes[i].is_ascii_digit() {
380 i += 1;
381 }
382 if i >= bytes.len() || bytes[i] != b'k' {
383 continue;
384 }
385
386 let before_ok = start == 0 || !bytes[start - 1].is_ascii_alphanumeric();
387 let after_ok = i + 1 >= bytes.len() || !bytes[i + 1].is_ascii_alphanumeric();
388 if !before_ok || !after_ok {
389 continue;
390 }
391
392 if let Ok(kilo_tokens) = model_lower[start..i].parse::<u32>()
393 && (8..=1024).contains(&kilo_tokens)
394 {
395 return Some(kilo_tokens.saturating_mul(1000));
396 }
397 } else {
398 i += 1;
399 }
400 }
401 None
402 }
403
404 /// Derive a compaction token threshold from model context and a caller-supplied
405 /// percentage.
406 #[must_use]
407 #[cfg(any(test, feature = "test-support"))]
408 pub fn compaction_threshold_for_model_at_percent(model: &str, percent: f64) -> usize {
409 let Some(window) = context_window_for_model(model) else {
410 return DEFAULT_COMPACTION_TOKEN_THRESHOLD;
411 };
412
413 let percent = percent.clamp(10.0, 100.0);
414 let threshold = (f64::from(window) * percent / 100.0).round();
415 let threshold = if threshold.is_finite() && threshold > 0.0 {
416 threshold as u64
417 } else {
418 u64::from(window) * u64::from(COMPACTION_THRESHOLD_PERCENT) / 100
419 };
420 usize::try_from(threshold).unwrap_or(DEFAULT_COMPACTION_TOKEN_THRESHOLD)
421 }
422
423 /// Whether auto-compaction should be enabled when the user did not explicitly
424 /// configure it. Known model windows default automatic continuity on; an
425 /// explicit `auto_compact = false` remains authoritative at the call sites.
426 #[must_use]
427 #[cfg(test)]
428 pub fn auto_compact_default_for_model(model: &str) -> bool {
429 context_window_for_model(model).is_some()
430 }
431
432 // === Streaming Structures ===
433
434 #[allow(dead_code)]
435 #[derive(Debug, Deserialize, Clone)]
436 #[serde(tag = "type")]
437 /// Streaming event types for SSE responses.
438 pub enum StreamEvent {
439 /// Local pre-stream receipt: the provider request was sent with a reduced
440 /// tool surface. This is not provider SSE and must not count as content.
441 #[serde(rename = "tool_projection_warning")]
442 ToolProjectionWarning {
443 provider: String,
444 omitted_tool_names: Vec<String>,
445 omitted_tool_count: usize,
446 },
447 #[serde(rename = "message_start")]
448 MessageStart { message: MessageResponse },
449 #[serde(rename = "content_block_start")]
450 ContentBlockStart {
451 index: u32,
452 content_block: ContentBlockStart,
453 },
454 #[serde(rename = "content_block_delta")]
455 ContentBlockDelta { index: u32, delta: Delta },
456 #[serde(rename = "content_block_stop")]
457 ContentBlockStop { index: u32 },
458 #[serde(rename = "message_delta")]
459 MessageDelta {
460 delta: MessageDelta,
461 usage: Option<Usage>,
462 },
463 #[serde(rename = "message_stop")]
464 MessageStop,
465 #[serde(rename = "ping")]
466 Ping,
467 /// Anthropic SSE error event (#3014).
468 #[serde(rename = "error")]
469 Error { error: serde_json::Value },
470 }
471
472 #[allow(dead_code)]
473 #[derive(Debug, Deserialize, Clone)]
474 #[serde(tag = "type")]
475 /// Content block types used in streaming starts.
476 pub enum ContentBlockStart {
477 #[serde(rename = "text")]
478 Text { text: String },
479 #[serde(rename = "thinking")]
480 Thinking { thinking: String },
481 #[serde(rename = "tool_use")]
482 ToolUse {
483 id: String,
484 name: String,
485 input: serde_json::Value, // usually empty or partial
486 #[serde(skip_serializing_if = "Option::is_none")]
487 caller: Option<ToolCaller>,
488 /// Google thought signature, when the first streaming chunk of this
489 /// tool call carried `extra_content.google.thought_signature`.
490 #[serde(skip_serializing_if = "Option::is_none")]
491 thought_signature: Option<String>,
492 },
493 #[serde(rename = "server_tool_use")]
494 ServerToolUse {
495 id: String,
496 name: String,
497 input: serde_json::Value,
498 },
499 }
500
501 // Variant names match legacy streaming spec, suppressing style warning
502 #[allow(clippy::enum_variant_names)]
503 #[derive(Debug, Deserialize, Clone)]
504 #[serde(tag = "type")]
505 /// Delta events emitted during streaming responses.
506 pub enum Delta {
507 #[serde(rename = "text_delta")]
508 TextDelta { text: String },
509 #[serde(rename = "thinking_delta")]
510 ThinkingDelta { thinking: String },
511 #[serde(rename = "input_json_delta")]
512 InputJsonDelta { partial_json: String },
513 /// Anthropic signed-thinking signature delta (#3014); arrives at the end
514 /// of a thinking block on the native Messages stream.
515 #[serde(rename = "signature_delta")]
516 SignatureDelta { signature: String },
517 /// Opaque Responses reasoning continuity, attached only when the provider
518 /// returns an encrypted item on the exact originating route.
519 #[serde(rename = "reasoning_state_delta")]
520 ReasoningStateDelta { state: OpaqueReasoningState },
521 }
522
523 #[allow(dead_code)]
524 #[derive(Debug, Deserialize, Clone)]
525 /// Delta payload for message-level updates.
526 pub struct MessageDelta {
527 pub stop_reason: Option<String>,
528 pub stop_sequence: Option<String>,
529 }
530
531 #[cfg(test)]
532 mod tests {
533 use super::*;
534 use std::any::TypeId;
535
536 /// #6032: `model_supports_reasoning` consults the catalog before its
537 /// hand-maintained pile, so a literal arm that duplicates a catalog row is
538 /// unreachable. These 27 arms were exactly that and were deleted. They must
539 /// keep answering `true` from the catalog *alone* — if a row is ever
540 /// dropped, this fails loudly here rather than silently reverting them to
541 /// "reasoning not expected", which leaks their `reasoning_content` into
542 /// ordinary prose (#6044).
543 #[test]
544 fn unknown_reasoning_capability_is_observable() {
545 assert_eq!(
546 model_reasoning_capability("not-a-real-model-xyz"),
547 None,
548 "unknown must not collapse to false at this layer"
549 );
550 assert!(
551 !model_supports_reasoning("not-a-real-model-xyz"),
552 "legacy bool wrapper still defaults unknown to false"
553 );
554 assert_eq!(model_reasoning_capability("kimi-for-coding"), Some(true));
555 }
556
557 #[test]
558 fn catalog_alone_covers_the_models_removed_from_the_heuristic_pile() {
559 let removed = [
560 "claude-opus-4-8",
561 "claude-opus-5",
562 "claude-sonnet-4-6",
563 "claude-sonnet-5",
564 "claude-fable-5",
565 "gpt-5-codex",
566 "gpt-5.3-codex",
567 "trinity-mini",
568 "trinity-large-thinking",
569 "moonshotai/kimi-k2.7-code",
570 "kimi-k2.7-code",
571 "minimax/minimax-m3",
572 "minimax/minimax-m2.7",
573 "minimax-m2.7",
574 "qwen/qwen3.6-flash",
575 "mimo-v2.5",
576 "mimo-v2.5-pro",
577 "mimo-v2.5-pro-ultraspeed",
578 "z-ai/glm-5.2",
579 "z-ai/glm-5.3",
580 "z-ai/glm-5.3-flash",
581 "glm-5.2",
582 "glm-5.3",
583 "glm-5.3-flash",
584 "muse-spark-1.1",
585 "muse-spark-1.2",
586 "muse-spark-1.2-contributor",
587 // Part 2: cited qwen3.x / Kimi coding-route arms moved into the
588 // bundled catalog (Alibaba Cloud Model Studio deep-thinking docs;
589 // #3016 plus the 2026 K2.7 update).
590 "kimi-for-coding",
591 "kimi-for-coding-highspeed",
592 "kimi-k2.5",
593 "kimi-k2.6",
594 "qwen3.5-flash",
595 "qwen3.5-plus",
596 "qwen3.6-flash",
597 "qwen3.6-plus",
598 "qwen3.7-max",
599 "qwen3.7-plus",
600 "qwen3.8-flash",
601 "qwen3.8-max",
602 "qwen3.8-max-preview",
603 ];
604 for model in removed {
605 assert_eq!(
606 codewhale_config::catalog::reviewed::intrinsic_model(model)
607 .and_then(|row| row.reasoning),
608 Some(true),
609 "{model} no longer has a catalog row, but its heuristic arm was \
610 deleted in #6032 — restore the row, or put the arm back"
611 );
612 assert!(
613 model_supports_reasoning(model),
614 "{model} must still classify as reasoning-capable"
615 );
616 }
617 }
618
619 /// #6032 part 3: the eight literal arms deleted in favor of the bundled
620 /// Models.dev snapshot. Each id's `reasoning` bool is sourced there (all
621 /// `true`), so they must keep answering without a heuristic arm — if a
622 /// row is dropped from the asset, this fails loudly instead of silently
623 /// reverting them to "reasoning not expected" (#6044).
624 #[test]
625 fn models_dev_bundled_asset_covers_the_arms_deleted_from_the_heuristic_pile() {
626 let catalog = codewhale_config::catalog::bundled_models_dev_catalog();
627 let deleted = [
628 "qwen/qwen3.8-flash",
629 "qwen/qwen3.6-35b-a3b",
630 "qwen/qwen3.6-plus",
631 "qwen/qwen3.7-plus",
632 "glm-5.1",
633 "grok-4.6",
634 "grok-4.5",
635 "grok-4.3",
636 ];
637 for model in deleted {
638 assert_eq!(
639 catalog.reasoning_support(model),
640 Some(true),
641 "{model} lost its bundled Models.dev row — restore its reviewed \
642 source row (#6032)"
643 );
644 assert!(
645 model_supports_reasoning(model),
646 "{model} must still classify as reasoning-capable"
647 );
648 }
649 }
650
651 /// #6032 coverage guard: exact reviewed facts and documented snapshot
652 /// contracts classify known models. Mistral's documented latest IDs now
653 /// have explicit reviewed reasoning facts; their family names and private
654 /// model names do not inherit those facts through a prefix heuristic.
655 #[test]
656 fn remaining_reasoning_heuristic_arms_are_unsourced_in_models_dev_bundled_asset() {
657 // Known facts are explicit data. Family resemblance is not a sourced
658 // capability, while the documented date-snapshot contract remains exact.
659 for model in [
660 "kimi-k2.9",
661 "magistral-medium",
662 "claude-private-server",
663 "gpt-5.6-private",
664 ] {
665 assert_eq!(model_reasoning_capability(model), None, "{model}");
666 }
667 for model in [
668 "mistral-medium-latest",
669 "mistral-small-latest",
670 "gpt-5.5-2026-06-01",
671 "gpt-5.1-codex-max",
672 "codex-gpt-5.5-preview",
673 "chatgpt-gpt-5.5",
674 ] {
675 assert_eq!(model_reasoning_capability(model), Some(true), "{model}");
676 }
677 }
678
679 #[test]
680 fn output_limit_stop_reason_accepts_provider_aliases_only() {
681 for reason in [
682 "length",
683 "max_tokens",
684 "max_output_tokens",
685 " MAX_TOKENS ",
686 "incomplete:max_output_tokens",
687 ] {
688 assert!(is_output_limit_stop_reason(Some(reason)), "{reason}");
689 }
690 for reason in [None, Some("end_turn"), Some("tool_use"), Some("")] {
691 assert!(!is_output_limit_stop_reason(reason), "{reason:?}");
692 }
693 }
694
695 #[test]
696 fn incomplete_stop_reason_never_accepts_unknown_responses_failures() {
697 assert!(is_incomplete_stop_reason(Some("incomplete:content_filter")));
698 assert!(is_incomplete_stop_reason(Some("content_filter")));
699 assert!(is_incomplete_stop_reason(Some(
700 "model_context_window_exceeded"
701 )));
702 assert!(is_incomplete_stop_reason(Some("max_tokens")));
703 assert!(!is_incomplete_stop_reason(Some("end_turn")));
704 assert_eq!(
705 stop_reason_detail(Some("incomplete:content_filter")),
706 "content_filter"
707 );
708 }
709
710 #[test]
711 fn historical_tui_request_path_is_the_core_request_type() {
712 assert_eq!(
713 TypeId::of::<MessageRequest>(),
714 TypeId::of::<codewhale_core::request::MessageRequest>()
715 );
716
717 let via_tui_path = MessageRequest {
718 model: "model".to_string(),
719 messages: vec![],
720 max_tokens: 1024,
721 system: None,
722 tools: None,
723 tool_choice: None,
724 metadata: None,
725 thinking: None,
726 reasoning_effort: None,
727 stream: Some(true),
728 temperature: None,
729 top_p: None,
730 };
731 let via_core_path: codewhale_core::request::MessageRequest = via_tui_path.clone();
732 assert_eq!(
733 serde_json::to_vec(&via_tui_path).expect("serialize TUI path"),
734 serde_json::to_vec(&via_core_path).expect("serialize core path")
735 );
736 }
737
738 #[test]
739 fn interrupted_assistant_role_round_trips_as_distinct_session_item() {
740 let message = Message {
741 role: Role::InterruptedAssistant,
742 content: vec![ContentBlock::Text {
743 text: "partial output".to_string(),
744 cache_control: None,
745 }],
746 };
747 let encoded = serde_json::to_string(&message).expect("message should serialize");
748 let decoded: Message = serde_json::from_str(&encoded).expect("message should deserialize");
749 assert_eq!(decoded, message);
750 assert_ne!(decoded.role, "assistant");
751 }
752
753 #[test]
754 fn unrecognized_deepseek_models_do_not_inherit_sibling_metadata() {
755 for model in [
756 "deepseek-v4-flash-20260423",
757 "deepseek-v4-pro-20260423",
758 "deepseek-coder",
759 "deepseek-v3.2-0324",
760 "deepseek-v4.1-flash-expires-on-0910",
761 ] {
762 assert_eq!(context_window_for_model(model), None, "{model}");
763 }
764 assert!(!model_supports_reasoning(
765 "deepseek-v4.1-flash-expires-on-0910"
766 ));
767 }
768
769 #[test]
770 fn deepseek_v4_models_map_to_1m_context_window() {
771 assert_eq!(
772 context_window_for_model("deepseek-v4-pro"),
773 Some(DEEPSEEK_V4_CONTEXT_WINDOW_TOKENS)
774 );
775 assert_eq!(
776 context_window_for_model("deepseek-v4-flash"),
777 Some(DEEPSEEK_V4_CONTEXT_WINDOW_TOKENS)
778 );
779 assert_eq!(
780 context_window_for_model("deepseek-ai/deepseek-v4-pro"),
781 Some(DEEPSEEK_V4_CONTEXT_WINDOW_TOKENS)
782 );
783 }
784
785 #[test]
786 fn deepseek_v4_output_caps_require_exact_catalog_metadata() {
787 for model in [
788 "deepseek-v4.1-flash-expires-on-0910",
789 "deepseek-v4.1-flash",
790 "deepseek-v4.1-pro",
791 "vendor/deepseek-v4.1-flash",
792 "deepseek-v4-flash-vendor",
793 ] {
794 assert!(
795 codewhale_config::catalog::reviewed::intrinsic_model(model).is_none(),
796 "{model}"
797 );
798 assert_eq!(max_output_tokens_for_model(model), None, "{model}");
799 }
800
801 for model in [
802 "deepseek-v4-flash",
803 "deepseek-v4-pro",
804 "deepseek-v4-flash-vision-exp",
805 "DEEPSEEK-V4-FLASH",
806 ] {
807 assert_eq!(max_output_tokens_for_model(model), Some(384_000), "{model}");
808 }
809 }
810
811 #[test]
812 fn recent_openrouter_large_models_have_static_windows() {
813 for (model, expected_window) in [
814 ("arcee-ai/trinity-large-thinking", 262_144),
815 ("trinity-large-thinking", 262_144),
816 (concat!("qwen/", "qwen3.8-flash"), 1_000_000),
817 (concat!("qwen/", "qwen3.6-flash"), 1_000_000),
818 (concat!("qwen/", "qwen3.6-35b-a3b"), 262_144),
819 (concat!("qwen/", "qwen3.6-max-preview"), 262_144),
820 (concat!("qwen/", "qwen3.6-plus"), 1_000_000),
821 (concat!("xiaomi/", "mimo-v2.5-pro"), 1_000_000),
822 ("mimo-v2.5-pro", 1_000_000),
823 ("mimo-v2.5-pro-ultraspeed", 1_000_000),
824 ("mimo-v2.5", 1_000_000),
825 ("minimax/minimax-m3", 1_000_000),
826 ("minimax/minimax-m2.7", 204_800),
827 ("moonshotai/kimi-k2.7-code", 262_144),
828 ("moonshotai/kimi-k2.6", 262_144),
829 ("google/gemma-4-31b-it", 262_144),
830 ("z-ai/glm-5.1", 202_752),
831 ("z-ai/glm-5.2", 1_000_000),
832 ("z-ai/glm-5.3", 1_000_000),
833 ("z-ai/glm-5.3-flash", 1_000_000),
834 ] {
835 assert_eq!(context_window_for_model(model), Some(expected_window));
836 assert!(model_supports_reasoning(model));
837 }
838 }
839
840 #[test]
841 fn openai_api_and_codex_models_have_verified_context_metadata() {
842 for model in ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"] {
843 assert_eq!(context_window_for_model(model), Some(1_050_000));
844 assert_eq!(max_output_tokens_for_model(model), Some(128_000));
845 assert!(model_supports_reasoning(model));
846 assert_eq!(
847 compaction_threshold_for_model_at_percent(model, 80.0),
848 840_000
849 );
850 }
851
852 for model in [
853 "gpt-5.5",
854 "gpt-5.5-pro",
855 "gpt-5.5-2026-04-23",
856 "gpt-5.5-pro-2026-04-23",
857 ] {
858 assert_eq!(context_window_for_model(model), Some(1_050_000));
859 assert_eq!(max_output_tokens_for_model(model), Some(128_000));
860 assert!(model_supports_reasoning(model));
861 assert_eq!(
862 compaction_threshold_for_model_at_percent(model, 80.0),
863 840_000
864 );
865 }
866
867 for model in [
868 "gpt-5-codex",
869 "gpt-5.1-codex",
870 "gpt-5.1-codex-mini",
871 "gpt-5.1-codex-max",
872 "gpt-5.2-codex",
873 "gpt-5.3-codex",
874 "codex-gpt-5.5",
875 "chatgpt-gpt-5.5",
876 "gpt-5.5-codex",
877 "gpt-5.5-codex-preview",
878 ] {
879 assert_eq!(context_window_for_model(model), Some(400_000));
880 assert_eq!(max_output_tokens_for_model(model), Some(128_000));
881 assert!(model_supports_reasoning(model));
882 assert_eq!(
883 compaction_threshold_for_model_at_percent(model, 80.0),
884 320_000
885 );
886 }
887
888 assert_eq!(context_window_for_model("gpt-5.5-nano"), None);
889 assert_eq!(max_output_tokens_for_model("gpt-5.5-nano"), None);
890 assert!(!model_supports_reasoning("gpt-5.5-nano"));
891 }
892
893 #[test]
894 fn anthropic_stepfun_and_sakana_limits_match_2026_07_09_audit() {
895 // Sonnet 4.6 output cap raised 64K -> 128K per
896 // https://platform.claude.com/docs/en/about-claude/models/overview;
897 // Haiku stays at 64K.
898 assert_eq!(
899 max_output_tokens_for_model("claude-sonnet-4-6"),
900 Some(128_000)
901 );
902 assert_eq!(
903 max_output_tokens_for_model("claude-haiku-4-5"),
904 Some(64_000)
905 );
906 // step-3.7-flash max output is third-party sourced (models.dev +
907 // Artificial Analysis; the official StepFun page is silent):
908 // https://models.dev/models/stepfun/step-3.7-flash/
909 assert_eq!(max_output_tokens_for_model("step-3.7-flash"), Some(256_000));
910 assert_eq!(context_window_for_model("step-3.7-flash"), Some(256_000));
911 // fugu-ultra limits are third-party sourced (Requesty; Sakana's own
912 // >272K price tier at https://console.sakana.ai/pricing confirms the
913 // context window exceeds 272K).
914 for model in ["fugu-ultra", "fugu-ultra-20260615"] {
915 assert_eq!(context_window_for_model(model), Some(1_000_000), "{model}");
916 assert_eq!(max_output_tokens_for_model(model), Some(131_000), "{model}");
917 }
918 }
919
920 #[test]
921 fn stepfun_current_coding_models_have_verified_metadata() {
922 assert_eq!(context_window_for_model("step-5-preview"), Some(1_000_000));
923 assert_eq!(
924 max_output_tokens_for_model("step-5-preview"),
925 Some(1_000_000)
926 );
927 for model in [
928 "step-5-preview",
929 "step-3.7-flash",
930 "step-3.5-flash",
931 "step-3.5-flash-2603",
932 ] {
933 assert!(model_supports_reasoning(model), "{model}");
934 }
935 for model in ["step-3.5-flash", "step-3.5-flash-2603"] {
936 assert_eq!(context_window_for_model(model), Some(256_000));
937 assert_eq!(max_output_tokens_for_model(model), None);
938 }
939 }
940
941 #[test]
942 fn claude_fable_5_and_sonnet_5_have_verified_metadata() {
943 // 1M context / 128K output per
944 // https://platform.claude.com/docs/en/about-claude/pricing (2026-07-09).
945 for model in ["claude-fable-5", "claude-sonnet-5"] {
946 assert_eq!(context_window_for_model(model), Some(1_000_000), "{model}");
947 assert_eq!(max_output_tokens_for_model(model), Some(128_000), "{model}");
948 assert!(model_supports_reasoning(model), "{model}");
949 }
950 }
951
952 #[test]
953 fn claude_opus_5_has_verified_metadata() {
954 // 1M context / 128K output, adaptive thinking, per
955 // https://platform.claude.com/docs/en/about-claude/models/overview
956 // (2026-08-17).
957 assert_eq!(context_window_for_model("claude-opus-5"), Some(1_000_000));
958 assert_eq!(max_output_tokens_for_model("claude-opus-5"), Some(128_000));
959 assert!(model_supports_reasoning("claude-opus-5"));
960 }
961
962 #[test]
963 fn kimi_k2_7_code_highspeed_shares_the_k2_7_code_limits() {
964 // https://platform.kimi.ai/docs/pricing/chat-k27-code (2026-08-17):
965 // same model as kimi-k2.7-code, 262,144 context.
966 for model in [
967 "kimi-k2.7-code-highspeed",
968 "moonshotai/kimi-k2.7-code-highspeed",
969 ] {
970 assert_eq!(context_window_for_model(model), Some(262_144), "{model}");
971 assert_eq!(max_output_tokens_for_model(model), Some(32_768), "{model}");
972 assert!(model_supports_reasoning(model), "{model}");
973 }
974 }
975
976 #[test]
977 fn gemini_api_models_have_documented_token_limits() {
978 // Every current Gemini API text model page lists 1,048,576 input /
979 // 65,536 output (verified 2026-08-17, see
980 // `known_context_window_for_model`).
981 for model in [
982 "gemini-3.7-flash",
983 "gemini-3.6-flash",
984 "gemini-3.5-flash",
985 "gemini-3.5-flash-lite",
986 "gemini-3.1-pro-preview",
987 "gemini-3-pro-preview",
988 "gemini-2.5-pro",
989 "gemini-2.5-flash",
990 ] {
991 assert_eq!(context_window_for_model(model), Some(1_048_576), "{model}");
992 assert_eq!(max_output_tokens_for_model(model), Some(65_536), "{model}");
993 }
994 }
995
996 #[test]
997 fn muse_spark_has_verified_context_and_reasoning_metadata() {
998 assert_eq!(context_window_for_model("muse-spark-1.1"), Some(1_000_000));
999 assert_eq!(max_output_tokens_for_model("muse-spark-1.1"), Some(32_000));
1000 assert!(model_supports_reasoning("muse-spark-1.1"));
1001 // Muse Spark 1.2 standard: 1M context, $1.25/$4.25 + $0.15 cache (Artificial Analysis).
1002 assert_eq!(context_window_for_model("muse-spark-1.2"), Some(1_000_000));
1003 assert_eq!(max_output_tokens_for_model("muse-spark-1.2"), Some(32_000));
1004 assert!(model_supports_reasoning("muse-spark-1.2"));
1005 // Contributor tier: same model/limits, ~12×/21× cheaper in exchange for training-data opt-in.
1006 assert_eq!(
1007 context_window_for_model("muse-spark-1.2-contributor"),
1008 Some(1_000_000)
1009 );
1010 assert_eq!(
1011 max_output_tokens_for_model("muse-spark-1.2-contributor"),
1012 Some(32_000)
1013 );
1014 assert!(model_supports_reasoning("muse-spark-1.2-contributor"));
1015 }
1016
1017 #[test]
1018 fn modelstudio_qwen38_max_is_1m_context_not_128k() {
1019 // Owner Token Plan console + curated catalog (2026-08-03). The 128K
1020 // figure is max output, not the window — never collapse them.
1021 for model in ["qwen3.8-max", "qwen3.8-max-preview"] {
1022 assert_eq!(context_window_for_model(model), Some(1_000_000), "{model}");
1023 assert_eq!(max_output_tokens_for_model(model), Some(131_072), "{model}");
1024 }
1025 }
1026
1027 #[test]
1028 fn openrouter_qwen38_flash_is_1m_context_with_128k_output() {
1029 // models.dev OpenRouter listing 2026-08-26: 1,000,000 / 131,072.
1030 // Both the namespaced wire id and the bare short id must resolve.
1031 for model in ["qwen/qwen3.8-flash", "qwen3.8-flash"] {
1032 assert_eq!(context_window_for_model(model), Some(1_000_000), "{model}");
1033 assert_eq!(max_output_tokens_for_model(model), Some(131_072), "{model}");
1034 assert!(model_supports_reasoning(model), "{model}");
1035 }
1036 }
1037
1038 #[test]
1039 fn modelstudio_bare_qwen_models_support_reasoning() {
1040 // Model Studio's deep-thinking docs: every qwen3.x family the Token /
1041 // Coding Plan catalogs carry is hybrid-thinking (reasoning_content on
1042 // the OpenAI dialect, thinking blocks on the Anthropic dialect).
1043 for model in [
1044 "qwen3.8-max",
1045 "qwen3.8-max-preview",
1046 "qwen3.7-max",
1047 "qwen3.7-plus",
1048 "qwen3.6-plus",
1049 "qwen3.6-flash",
1050 "qwen3.5-plus",
1051 "qwen3.5-flash",
1052 ] {
1053 assert!(model_supports_reasoning(model), "{model}");
1054 }
1055 }
1056
1057 #[test]
1058 fn model_metadata_catalog_override_flows_through_models_chokepoint() {
1059 let private=codewhale_config::models_dev::ModelsDevCatalog::parse_json(r#"{"providers":{"private":{"models":{"catalog-only-model":{"limit":{"context":777000,"output":55000},"reasoning":true}}}}}"#).unwrap();
1060 assert!(
1061 codewhale_config::catalog::reviewed::intrinsic_model_in(&private, "catalog-only-model")
1062 .is_none()
1063 );
1064 assert_eq!(context_window_for_model("catalog-only-model"), None);
1065 assert_eq!(max_output_tokens_for_model("catalog-only-model"), None);
1066 assert_eq!(model_reasoning_capability("catalog-only-model"), None);
1067 }
1068
1069 #[test]
1070 fn moonshot_native_kimi_ids_support_reasoning_including_coding_route() {
1071 // #3016: bare Moonshot ids (no moonshotai/ prefix) emit
1072 // reasoning_content; kimi-for-coding currently rides the K2.7 Code path.
1073 assert!(model_supports_reasoning("kimi-k2.7-code"));
1074 assert!(model_supports_reasoning("kimi-k2.6"));
1075 assert!(model_supports_reasoning("kimi-for-coding"));
1076 assert!(model_supports_reasoning("kimi-for-coding-highspeed"));
1077 assert!(model_supports_reasoning("kimi-k2.5"));
1078 }
1079
1080 #[test]
1081 fn xai_grok_models_have_static_context_metadata() {
1082 for (model, expected_window, supports_reasoning) in [
1083 ("grok-4.6", 500_000, true),
1084 ("grok-4.5", 500_000, true),
1085 ("grok-4.3", 1_000_000, true),
1086 ("grok-build", 512_000, true),
1087 ("grok-composer-2.5-fast", 200_000, false),
1088 ("grok-4.20-0309-reasoning", 2_000_000, true),
1089 ("grok-4.20-0309-non-reasoning", 2_000_000, false),
1090 ] {
1091 assert_eq!(context_window_for_model(model), Some(expected_window));
1092 assert_eq!(max_output_tokens_for_model(model), None);
1093 assert_eq!(model_supports_reasoning(model), supports_reasoning);
1094 }
1095 }
1096
1097 #[test]
1098 fn arcee_direct_models_preserve_verified_capabilities_only() {
1099 assert_eq!(
1100 context_window_for_model("trinity-large-preview"),
1101 Some(262_144)
1102 );
1103 assert!(!model_supports_reasoning("trinity-large-preview"));
1104 assert_eq!(context_window_for_model("trinity-mini"), Some(128_000));
1105 assert_eq!(max_output_tokens_for_model("trinity-mini"), None);
1106 assert!(model_supports_reasoning("trinity-mini"));
1107 }
1108
1109 #[test]
1110 fn qwen37_plus_and_inkling_reasoning_do_not_invent_limits() {
1111 for model in ["qwen/qwen3.7-plus", "thinkingmachines/inkling"] {
1112 assert_eq!(context_window_for_model(model), None, "{model}");
1113 assert_eq!(max_output_tokens_for_model(model), None, "{model}");
1114 assert!(model_supports_reasoning(model), "{model}");
1115 }
1116 }
1117
1118 #[test]
1119 fn recent_openrouter_large_models_have_known_output_caps() {
1120 assert_eq!(
1121 max_output_tokens_for_model("arcee-ai/trinity-large-thinking"),
1122 Some(262_144)
1123 );
1124 assert_eq!(
1125 max_output_tokens_for_model("trinity-large-thinking"),
1126 Some(262_144)
1127 );
1128 assert_eq!(
1129 max_output_tokens_for_model(concat!("qwen/", "qwen3.8-flash")),
1130 Some(131_072)
1131 );
1132 assert_eq!(
1133 max_output_tokens_for_model(concat!("qwen/", "qwen3.6-flash")),
1134 Some(65_536)
1135 );
1136 assert_eq!(
1137 max_output_tokens_for_model(concat!("qwen/", "qwen3.6-max-preview")),
1138 Some(65_536)
1139 );
1140 assert_eq!(
1141 max_output_tokens_for_model(concat!("qwen/", "qwen3.6-plus")),
1142 Some(65_536)
1143 );
1144 assert_eq!(
1145 max_output_tokens_for_model(concat!("xiaomi/", "mimo-v2.5-pro")),
1146 Some(131_072)
1147 );
1148 assert_eq!(max_output_tokens_for_model("mimo-v2.5-pro"), Some(131_072));
1149 assert_eq!(
1150 max_output_tokens_for_model("mimo-v2.5-pro-ultraspeed"),
1151 Some(131_072)
1152 );
1153 assert_eq!(max_output_tokens_for_model("mimo-v2.5"), Some(131_072));
1154 assert_eq!(
1155 max_output_tokens_for_model("minimax/minimax-m3"),
1156 Some(524_288)
1157 );
1158 assert_eq!(max_output_tokens_for_model("z-ai/glm-5.1"), Some(131_072));
1159 assert_eq!(max_output_tokens_for_model("z-ai/glm-5.2"), Some(131_072));
1160 assert_eq!(max_output_tokens_for_model("z-ai/glm-5.3"), Some(131_072));
1161 assert_eq!(
1162 max_output_tokens_for_model("z-ai/glm-5-turbo"),
1163 Some(131_072)
1164 );
1165 assert_eq!(max_output_tokens_for_model("glm-5-turbo"), Some(131_072));
1166 }
1167
1168 #[test]
1169 fn k3_route_ids_use_verified_contracts_not_legacy_128k() {
1170 // Open-platform K3 carries the verified 1M contract.
1171 assert_eq!(context_window_for_model("kimi-k3"), Some(1_048_576));
1172 assert_eq!(
1173 context_window_for_model("opencode-go/kimi-k3"),
1174 Some(1_048_576)
1175 );
1176 // Bare `k3` (Kimi Code membership) is plan-tier dependent, so it
1177 // keeps the documented safe floor — and must never fall through to
1178 // the 128K legacy default.
1179 assert_eq!(context_window_for_model("k3"), Some(262_144));
1180 assert_eq!(context_window_for_model("k3-256k"), Some(262_144));
1181 assert_eq!(max_output_tokens_for_model("k3"), Some(131_072));
1182 assert_eq!(max_output_tokens_for_model("k3-256k"), Some(131_072));
1183 assert_eq!(max_output_tokens_for_model("kimi-k3"), Some(131_072));
1184 // Never project max output as the context window.
1185 assert_ne!(
1186 context_window_for_model("k3"),
1187 max_output_tokens_for_model("k3")
1188 );
1189 assert_ne!(
1190 context_window_for_model("kimi-k3"),
1191 max_output_tokens_for_model("kimi-k3")
1192 );
1193 }
1194
1195 #[test]
1196 fn kimi_code_membership_ids_mirror_their_family_facts() {
1197 // The high-speed membership id rides the kimi-for-coding family
1198 // context fact (256K) and reasoning support via the same `kimi-`
1199 // native-id rule as `kimi-for-coding`. No client-side output ceiling
1200 // is claimed for the membership ids — the membership catalog is the
1201 // source of truth, so the generic lookup returns None.
1202 assert_eq!(
1203 context_window_for_model("kimi-for-coding-highspeed"),
1204 Some(262_144)
1205 );
1206 assert_eq!(
1207 max_output_tokens_for_model("kimi-for-coding-highspeed"),
1208 None
1209 );
1210 assert_eq!(max_output_tokens_for_model("kimi-for-coding"), None);
1211 assert!(model_supports_reasoning("kimi-for-coding-highspeed"));
1212 }
1213
1214 #[test]
1215 fn bare_provider_model_ids_mirror_vendor_prefixed_rows() {
1216 // Direct-provider routes (Moonshot, MiniMax, Z.ai) serve bare model
1217 // ids without the OpenRouter vendor prefix; both spellings must
1218 // resolve identical metadata (#1310 ride-along on #3023).
1219 for (model, expected_window) in [
1220 ("kimi-k3", 1_048_576),
1221 ("kimi-k2.7-code", 262_144),
1222 ("kimi-k2.6", 262_144),
1223 ("minimax-m3", 1_000_000),
1224 ("minimax-m2.7", 204_800),
1225 ("minimax-m2.5-highspeed", 204_800),
1226 ("minimax-m2", 204_800),
1227 ("glm-5.1", 202_752),
1228 ("glm-5.2", 1_000_000),
1229 // Inherited from glm-5.2 pending official Z.ai release metadata.
1230 ("glm-5.3", 1_000_000),
1231 ("glm-5.3-flash", 1_000_000),
1232 ("glm-5-turbo", 202_752),
1233 ] {
1234 assert_eq!(context_window_for_model(model), Some(expected_window));
1235 assert!(model_supports_reasoning(model));
1236 }
1237 assert_eq!(context_window_for_model("kimi-for-coding"), Some(262_144));
1238 assert!(model_supports_reasoning("kimi-for-coding"));
1239 assert_eq!(context_window_for_model("glm-5v-turbo"), Some(202_752));
1240 assert!(!model_supports_reasoning("glm-5v-turbo"));
1241 // GLM-5-Turbo is a fast text sibling (distinct from the glm-5v-turbo
1242 // vision model): same compact window as 5.1 but reasoning-capable.
1243 assert_eq!(context_window_for_model("z-ai/glm-5-turbo"), Some(202_752));
1244 assert!(model_supports_reasoning("z-ai/glm-5-turbo"));
1245 assert_eq!(
1246 codewhale_config::catalog::reviewed::intrinsic_model("kimi-k2.7-code")
1247 .and_then(|row| row.generation_default.or(row.max_output)),
1248 Some(32_768)
1249 );
1250 assert_eq!(max_output_tokens_for_model("kimi-k2.7-code"), Some(32_768));
1251 assert_eq!(max_output_tokens_for_model("kimi-k2.6"), Some(32_768));
1252 assert_eq!(max_output_tokens_for_model("kimi-for-coding"), None);
1253 assert_eq!(max_output_tokens_for_model("kimi-k3"), Some(131_072));
1254 assert_eq!(max_output_tokens_for_model("minimax-m3"), Some(524_288));
1255 assert_eq!(max_output_tokens_for_model("glm-5.1"), Some(131_072));
1256 assert_eq!(max_output_tokens_for_model("glm-5.2"), Some(131_072));
1257 assert_eq!(max_output_tokens_for_model("glm-5.3"), Some(131_072));
1258 assert_eq!(max_output_tokens_for_model("glm-5.3-flash"), Some(131_072));
1259 }
1260
1261 #[test]
1262 fn deepseek_models_with_k_suffix_use_hint() {
1263 assert_eq!(context_window_for_model("deepseek-v3.2-32k"), Some(32_000));
1264 assert_eq!(
1265 context_window_for_model("deepseek-v3.2-256k-preview"),
1266 Some(256_000)
1267 );
1268 assert_eq!(context_window_for_model("deepseek-v3.2-2k-preview"), None);
1269 }
1270
1271 /// 2026-10-05: custom gateways serve V4 under snapshot and variant ids
1272 /// the reviewed catalog does not enumerate (`deepseek-v4-pro-0813`,
1273 /// `deepseek-v4-flash-vision`). They denote the same rows as their base
1274 /// ids and must inherit the base facts instead of falling back to the
1275 /// 128K unknown shape.
1276 #[test]
1277 fn deepseek_v4_snapshot_and_variant_ids_inherit_reviewed_facts() {
1278 for model in [
1279 "deepseek-v4-pro-0813",
1280 "DeepSeek-V4-Pro-0813",
1281 "deepseek-v4-pro-2025-08-13",
1282 "deepseek-v4-flash-vision",
1283 "deepseek-v4-flash-vision-0813",
1284 ] {
1285 assert_eq!(context_window_for_model(model), Some(1_000_000), "{model}");
1286 assert_eq!(max_output_tokens_for_model(model), Some(384_000), "{model}");
1287 assert!(model_supports_reasoning(model), "{model}");
1288 }
1289 // A trailing number that is not a recognized date stamp, an
1290 // eight-digit compact stamp, or an unknown base, stays unknown: the
1291 // resolver never invents a row.
1292 for model in [
1293 "not-a-model-0813",
1294 "deepseek-v4-pro-9913",
1295 "deepseek-v4-pro-08130",
1296 "deepseek-v4-pro-x0813",
1297 "deepseek-v4-pro-20250813",
1298 ] {
1299 assert_eq!(context_window_for_model(model), None, "{model}");
1300 assert_eq!(model_reasoning_capability(model), None, "{model}");
1301 }
1302 // Snapshot resolution must not relabel a DeepSeek row as an OpenAI
1303 // reasoning model, while the OpenAI snapshot contract keeps working.
1304 assert!(!model_is_openai_reasoning_family("deepseek-v4-pro-0813"));
1305 assert!(model_is_openai_reasoning_family("gpt-5.5-2026-06-01"));
1306 }
1307
1308 /// Multi-byte model ids must not panic the byte-indexed date scan: the
1309 /// `-YYYY-MM-DD` layer splits at `len - 10`, which need not be a UTF-8
1310 /// char boundary (review finding from the PR #6 review).
1311 #[test]
1312 fn non_ascii_ids_do_not_panic_the_snapshot_scan() {
1313 for model in ["模型模型模型模型", "ローカルモデル-2026", "モデル-0813"] {
1314 assert_eq!(context_window_for_model(model), None, "{model}");
1315 assert_eq!(model_reasoning_capability(model), None, "{model}");
1316 }
1317 }
1318
1319 #[test]
1320 fn compaction_threshold_scales_with_context_window() {
1321 assert_eq!(
1322 compaction_threshold_for_model_at_percent("deepseek-v3.2-128k", 80.0),
1323 102_400
1324 );
1325 // v0.8.11 (#664): unknown-model fallback also resolves to 80% of
1326 // `LEGACY_DEEPSEEK_CONTEXT_WINDOW_TOKENS` (128K legacy DeepSeek
1327 // fallback) — same late-trigger discipline as the V4 path. Was
1328 // `50_000` pre-v0.8.11; that hardcoded value compacted at ~5% of a
1329 // 1M window when model detection silently fell through, which is
1330 // exactly the prefix-cache-burning behaviour we're getting away from.
1331 assert_eq!(
1332 compaction_threshold_for_model_at_percent("unknown-model", 80.0),
1333 102_400
1334 );
1335 }
1336
1337 #[test]
1338 fn compaction_scales_for_deepseek_v4_1m_context() {
1339 assert_eq!(
1340 compaction_threshold_for_model_at_percent("deepseek-v4-pro", 80.0),
1341 800_000
1342 );
1343 }
1344
1345 #[test]
1346 fn compaction_threshold_honors_configured_percent() {
1347 assert_eq!(
1348 compaction_threshold_for_model_at_percent("deepseek-v4-pro", 75.0),
1349 750_000
1350 );
1351 assert_eq!(
1352 compaction_threshold_for_model_at_percent("trinity-large-thinking", 80.0),
1353 209_715
1354 );
1355 }
1356
1357 #[test]
1358 fn auto_compaction_defaults_on_for_known_supported_model_windows() {
1359 assert!(auto_compact_default_for_model("trinity-large-thinking"));
1360 assert!(auto_compact_default_for_model("deepseek-v3.2-128k"));
1361 assert!(auto_compact_default_for_model("deepseek-v4-pro"));
1362 assert!(auto_compact_default_for_model("mimo-v2.5-pro"));
1363 assert!(!auto_compact_default_for_model("unknown-model"));
1364 }
1365 }
1366
1366 lines RUST