返回 CodeWhale
route_budget.rs
根目录 / crates / tui / src / route_budget.rs
1 use codewhale_config::route::RouteLimits;
2
3 use crate::config::{ProviderKind, provider_capability};
4 use crate::context_budget::ContextBudget;
5 use codewhale_models::{DEFAULT_COMPACTION_TOKEN_THRESHOLD, context_window_for_model};
6
7 /// Safe ordinary API request cap across provider routes.
8 const API_MAX_OUTPUT_TOKENS: u32 = 65_536;
9
10 /// Preserve only route limits that came from a concrete offering.
11 #[must_use]
12 pub(crate) fn known_route_limits(limits: RouteLimits) -> Option<RouteLimits> {
13 limits.has_known_limit().then_some(limits)
14 }
15
16 /// Whether this exact transport can represent an explicit output allowance.
17 /// Codex OAuth Responses rejects the field; other supported dialects carry it.
18 pub(crate) fn route_supports_output_token_limit(
19 provider: ProviderKind,
20 protocol: codewhale_config::route::RequestProtocol,
21 ) -> bool {
22 !(provider == ProviderKind::OpenaiCodex
23 && protocol == codewhale_config::route::RequestProtocol::Responses)
24 }
25
26 pub(crate) fn effective_max_output_tokens_for_turn(
27 provider: ProviderKind,
28 model: &str,
29 route_limits: Option<RouteLimits>,
30 allowance: Option<std::num::NonZeroU32>,
31 ) -> u32 {
32 let ceiling = effective_max_output_tokens_for_route(provider, model, route_limits);
33 allowance.map_or(ceiling, |allowance| ceiling.min(allowance.get()))
34 }
35
36 /// Context window for a resolved runtime route.
37 ///
38 /// Route/offering facts win when known; otherwise this falls back to the
39 /// existing provider+model capability matrix so startup and custom/local
40 /// routes keep their previous conservative behavior.
41 #[must_use]
42 pub(crate) fn route_context_window_tokens(
43 provider: ProviderKind,
44 model: &str,
45 route_limits: Option<RouteLimits>,
46 ) -> u32 {
47 route_limits
48 .and_then(|limits| limits.context_tokens)
49 .and_then(|tokens| u32::try_from(tokens).ok())
50 .filter(|tokens| *tokens > 0)
51 .unwrap_or_else(|| provider_capability(provider, model).context_window)
52 }
53
54 /// Share of the route's context window (in estimated characters) one tool
55 /// result may occupy inline.
56 const TOOL_OUTPUT_INLINE_WINDOW_PERCENT: u64 = 3;
57 /// Inline ceiling for one tool result when the route is unknown, and the cap
58 /// for every route unless the operator opted into a larger budget.
59 const TOOL_OUTPUT_INLINE_MAX_CHARS: usize = 100_000;
60 /// Hard ceiling on an operator-raised inline tool-result budget (#5367).
61 const TOOL_OUTPUT_INLINE_OPT_IN_MAX_CHARS: usize = 2 * 1024 * 1024;
62
63 /// The one budget for how much of a tool result the model sees inline (#6508).
64 ///
65 /// Every tool result is measured against this number: 3% of the route's
66 /// context window (at about 4 characters per token), capped at 100,000
67 /// characters, or the full cap when the window is unknown. An opt-in
68 /// `tool_result_max_bytes` workshop setting may raise it, up to 2 MiB.
69 /// Anything past it is cut only after the raw output has been saved as a
70 /// session artifact that `retrieve_tool_result` can read back.
71 #[must_use]
72 pub(crate) fn route_inline_char_budget(window_tokens: Option<u32>) -> usize {
73 route_inline_char_budget_with_raise(
74 window_tokens,
75 crate::tools::large_output_router::WorkshopConfig::active_tool_result_max_bytes(),
76 )
77 }
78
79 #[must_use]
80 fn route_inline_char_budget_with_raise(window_tokens: Option<u32>, raise: Option<usize>) -> usize {
81 let base = window_tokens
82 .filter(|tokens| *tokens > 0)
83 .map(|tokens| {
84 let chars = u64::from(tokens)
85 .saturating_mul(4)
86 .saturating_mul(TOOL_OUTPUT_INLINE_WINDOW_PERCENT)
87 / 100;
88 usize::try_from(chars).unwrap_or(TOOL_OUTPUT_INLINE_MAX_CHARS)
89 })
90 .unwrap_or(TOOL_OUTPUT_INLINE_MAX_CHARS)
91 .clamp(1, TOOL_OUTPUT_INLINE_MAX_CHARS);
92 match raise {
93 Some(bytes) if bytes > base => bytes.min(TOOL_OUTPUT_INLINE_OPT_IN_MAX_CHARS),
94 _ => base,
95 }
96 }
97
98 /// [`route_inline_char_budget`] for a resolved provider/model route.
99 #[must_use]
100 pub(crate) fn route_inline_char_budget_for_route(
101 provider: ProviderKind,
102 model: &str,
103 route_limits: Option<RouteLimits>,
104 ) -> usize {
105 route_inline_char_budget(Some(route_context_window_tokens(
106 provider,
107 model,
108 route_limits,
109 )))
110 }
111
112 /// Provider/offering output cap, when the resolved route reports one.
113 #[must_use]
114 pub(crate) fn route_output_limit_tokens(route_limits: Option<RouteLimits>) -> Option<u32> {
115 route_limits
116 .and_then(|limits| limits.output_tokens)
117 .and_then(|tokens| u32::try_from(tokens).ok())
118 .filter(|tokens| *tokens > 0)
119 }
120
121 /// Provider/offering input cap, when the resolved route reports one.
122 #[must_use]
123 pub(crate) fn route_input_limit_tokens(route_limits: Option<RouteLimits>) -> Option<u32> {
124 route_limits
125 .and_then(|limits| limits.input_tokens)
126 .and_then(|tokens| u32::try_from(tokens).ok())
127 .filter(|tokens| *tokens > 0)
128 }
129
130 /// Explicit operator request cap, when configured.
131 ///
132 /// Keep this separate from catalogue/default resolution: a published maximum
133 /// is a ceiling, while these environment variables are an actual request from
134 /// the operator. Route/window validation still clamps the value before it is
135 /// sent.
136 #[must_use]
137 fn explicit_max_output_tokens_override() -> Option<u32> {
138 match std::env::var("CODEWHALE_MAX_OUTPUT_TOKENS") {
139 Ok(raw) if !raw.trim().is_empty() => {
140 // A non-blank canonical value is authoritative. Invalid/zero
141 // values deliberately fall back to the safe automatic default;
142 // they must not silently activate a stale legacy setting.
143 return raw.trim().parse::<u32>().ok().filter(|tokens| *tokens > 0);
144 }
145 Ok(_) | Err(_) => {}
146 }
147 std::env::var("DEEPSEEK_MAX_OUTPUT_TOKENS")
148 .ok()
149 .and_then(|raw| raw.trim().parse::<u32>().ok())
150 .filter(|tokens| *tokens > 0)
151 }
152
153 /// Effective `max_tokens` for a model before provider/route caps are applied.
154 #[must_use]
155 pub(crate) fn effective_max_output_tokens(model: &str) -> u32 {
156 if let Some(tokens) = explicit_max_output_tokens_override() {
157 return tokens;
158 }
159
160 // A documented catalogue value is a capability ceiling, not necessarily a
161 // sensible default request size. In particular, DeepSeek V4 advertises a
162 // 384K maximum. Treating that maximum as the default made Codewhale ask a
163 // 262K/327K self-hosted route for almost its whole context as output before
164 // it had counted a single input token (#5516/#5518). Keep documented
165 // ceilings through the normal compatibility intersection, but automatic
166 // requests start at the ordinary 64K cap unless the operator explicitly
167 // overrides it. A maximum describes what a provider may allow, not what
168 // every response should reserve by default.
169 //
170 // Provenance for the ceiling (deepseek-v4-flash/pro: 384_000 output):
171 // - models_dev.bundled.json documents limit.output = 384000.
172 // - The DS4 provider contract corroborates 384K
173 // (crates/config/src/model_reference.rs pins max_output 384_000 / "384K").
174 // - Official DeepSeek API docs confirm the model ids (deepseek-v4-flash ->
175 // V4-Flash-0731, deepseek-v4-pro -> V4-Pro-0813) but do not publish the
176 // output ceiling in a machine-readable form; that number remains a
177 // catalogue-sourced value to re-verify against official docs when they
178 // publish one (#5373).
179 if let Some(documented) = codewhale_models::max_output_tokens_for_model(model) {
180 return documented.min(API_MAX_OUTPUT_TOKENS);
181 }
182
183 let window = context_window_for_model(model).unwrap_or(128_000);
184 (window / 2).min(API_MAX_OUTPUT_TOKENS)
185 }
186
187 /// Automatic allowance when an exact remote model has no output metadata.
188 /// This is policy, not a discovered provider limit. Reasoning and tool arguments
189 /// share this allowance; an 8K fallback truncated ordinary file writes after
190 /// reasoning consumed most of the response. Known route limits still constrain
191 /// requests, and an explicit operator setting may replace this fallback.
192 const UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS: u32 = API_MAX_OUTPUT_TOKENS;
193
194 /// Assumed output ceiling for an Anthropic-family model the catalogue does
195 /// not describe (#5440). The 64K Messages floor is real, but applying it to
196 /// an unknown model is an assumption about that model, not a documented
197 /// fact, so it clamps under an `unverified` label.
198 const ANTHROPIC_UNKNOWN_MAX_OUTPUT_TOKENS: u32 = 64_000;
199
200 /// Assumed output ceiling for the ChatGPT/Codex OAuth route, which publishes
201 /// no output ceiling of its own (#5440). Same clamp as the long-standing 4K
202 /// policy — relabeled, not revalued.
203 const CODEX_OAUTH_MAX_OUTPUT_TOKENS: u32 = 4_096;
204
205 /// Why a route's compatibility output ceiling has the value it does.
206 ///
207 /// Carried so a clamp is always attributable: "unknown" is only allowed to
208 /// mean "no clamp" when a route *truthfully publishes no ceiling*, never when
209 /// the catalogue simply has no row for the model.
210 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
211 pub(crate) enum OutputCeilingSource {
212 /// The static catalogue publishes an exact/conservative ceiling.
213 Documented(u32),
214 /// The route is known to publish no output maximum we can stand behind
215 /// (Kimi Code membership ids, operator-owned self-hosted engines). Unknown
216 /// stays unknown and nothing is clamped.
217 RouteDeclaredUnknown,
218 /// The catalogue has no row for this model. Fail closed to a conservative
219 /// ceiling rather than treating absence as permission.
220 Uncatalogued(u32),
221 /// The route publishes no ceiling we can stand behind, but a defensible
222 /// floor is still applied: an Anthropic-family model the catalogue does
223 /// not describe (64K Messages floor) and the Codex OAuth route (4K
224 /// policy). Clamping trades late provider failure for early truncation;
225 /// lying about why is not part of that trade, so receipts and pickers
226 /// must render this as an assumption, never as "documented" (#5440).
227 Unverified(u32),
228 }
229
230 impl OutputCeilingSource {
231 /// The ceiling to intersect a requested cap with, if any.
232 #[must_use]
233 pub(crate) const fn clamp_tokens(self) -> Option<u32> {
234 match self {
235 Self::Documented(tokens) | Self::Uncatalogued(tokens) | Self::Unverified(tokens) => {
236 Some(tokens)
237 }
238 Self::RouteDeclaredUnknown => None,
239 }
240 }
241
242 /// Stable provenance label, surfaced in exec stream metadata so a wrong
243 /// ceiling is visible in a receipt rather than requiring packet capture.
244 #[must_use]
245 pub(crate) const fn as_str(self) -> &'static str {
246 match self {
247 Self::Documented(_) => "documented",
248 Self::Uncatalogued(_) => "uncatalogued",
249 Self::RouteDeclaredUnknown => "route-declared",
250 Self::Unverified(_) => "unverified",
251 }
252 }
253 }
254
255 /// Whether an absent compatibility ceiling is a *declared* unknown for this
256 /// route, rather than a gap in the catalogue.
257 ///
258 /// Deliberately an allowlist. Everything not named here is uncatalogued and
259 /// gets the conservative ceiling.
260 #[must_use]
261 fn route_declares_unknown_output_ceiling(provider: ProviderKind, model: &str) -> bool {
262 match provider {
263 // Operator-owned engines: the local server, not this process, owns the
264 // output ceiling, and it is routinely far above any catalogue row.
265 ProviderKind::Ollama | ProviderKind::Sglang | ProviderKind::Vllm => true,
266 // Kimi Code membership ids publish their limits in the membership
267 // catalog rather than the static model catalogue.
268 ProviderKind::Moonshot => crate::config::is_kimi_code_membership_model(model),
269 _ => false,
270 }
271 }
272
273 /// Resolve the compatibility output ceiling for a route, with its provenance.
274 #[must_use]
275 pub(crate) fn output_ceiling_source(provider: ProviderKind, model: &str) -> OutputCeilingSource {
276 // #5440: two routes clamp to a number the route itself never documented.
277 // The clamps stay (see `OutputCeilingSource::Unverified`); the labels must
278 // not borrow the documented rung's authority.
279 if provider == ProviderKind::OpenaiCodex {
280 return OutputCeilingSource::Unverified(CODEX_OAUTH_MAX_OUTPUT_TOKENS);
281 }
282 if matches!(
283 provider,
284 ProviderKind::Anthropic | ProviderKind::MinimaxAnthropic | ProviderKind::Openmodel
285 ) && codewhale_models::max_output_tokens_for_model(model).is_none()
286 {
287 return OutputCeilingSource::Unverified(ANTHROPIC_UNKNOWN_MAX_OUTPUT_TOKENS);
288 }
289 provider_capability(provider, model).max_output.map_or_else(
290 || {
291 if route_declares_unknown_output_ceiling(provider, model) {
292 OutputCeilingSource::RouteDeclaredUnknown
293 } else {
294 OutputCeilingSource::Uncatalogued(UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS)
295 }
296 },
297 OutputCeilingSource::Documented,
298 )
299 }
300
301 /// Effective request output cap for a fully resolved provider/model route.
302 #[must_use]
303 pub(crate) fn effective_max_output_tokens_for_route(
304 provider: ProviderKind,
305 model: &str,
306 route_limits: Option<RouteLimits>,
307 ) -> u32 {
308 let window = route_context_window_tokens(provider, model, route_limits);
309 let compatibility_source = output_ceiling_source(provider, model);
310 let compatibility_cap = compatibility_source.clamp_tokens();
311 let route_cap = route_output_limit_tokens(route_limits);
312 // With a known route window and no published output limit, reserve a
313 // conservative part of that window. The model-only fallback reserved 64K even
314 // for a configured 32K Ollama route, leaving just 1K for input (#5820).
315 // The same squeeze hit the capability *fallback* window: an unknown local
316 // Ollama tag resolves to an 8K window with no route limits, the model-only
317 // 64K request clamped to 6K, and the input budget collapsed to 1K — every
318 // turn tripped emergency compaction before its first request (#6540). So a
319 // model-only request larger than half of whatever window is in force also
320 // yields to the window-relative reservation.
321 // Explicit requests and documented ceilings retain their existing rules.
322 let model_only_cap = effective_max_output_tokens(model);
323 let window_known = route_limits
324 .and_then(|limits| limits.context_tokens)
325 .is_some_and(|tokens| (1..=u64::from(u32::MAX)).contains(&tokens));
326 let requested_cap = if explicit_max_output_tokens_override().is_none()
327 && codewhale_models::max_output_tokens_for_model(model).is_none()
328 && route_cap.is_none()
329 && matches!(
330 compatibility_source,
331 OutputCeilingSource::RouteDeclaredUnknown | OutputCeilingSource::Uncatalogued(_)
332 )
333 && (window_known || model_only_cap > window / 2)
334 {
335 (window / 4).clamp(1, UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS)
336 } else {
337 model_only_cap
338 };
339 // Unknown means unknown only where a route *declares* it: membership ids
340 // such as the `kimi-for-coding` family, and operator-owned self-hosted
341 // engines. For those there is nothing to clamp against and the requested
342 // cap stands. A model the catalogue simply has no row for is not the same
343 // fact — absence keeps a labeled automatic allowance
344 // (see `output_ceiling_source`). A concrete route/offering maximum is the
345 // missing evidence for that exact route and may replace only the generic
346 // uncatalogued guess; known compatibility caps stay authoritative and are
347 // still intersected with any route maximum.
348 let cap = match (compatibility_source, route_cap) {
349 // A concrete route/offering maximum is evidence about this exact
350 // route. It therefore outranks the generic fallback that exists only
351 // because the static catalogue has no row for the wire id. With no
352 // route fact the conservative guess still applies, and the route fact
353 // can never raise the caller's requested cap.
354 (OutputCeilingSource::Uncatalogued(_), Some(route_cap)) => requested_cap.min(route_cap),
355 (OutputCeilingSource::Uncatalogued(_), None)
356 if explicit_max_output_tokens_override().is_some() =>
357 {
358 requested_cap
359 }
360 _ => {
361 let cap = compatibility_cap.map_or(requested_cap, |compat| requested_cap.min(compat));
362 route_cap.map_or(cap, |route_cap| cap.min(route_cap))
363 }
364 };
365 // Clamp against the effective route window even when it came from the
366 // capability fallback rather than an explicit offering. This keeps a
367 // suffix/config/catalog-derived small window from ever receiving a request
368 // cap larger than the window itself.
369 u32::try_from(ContextBudget::new(u64::from(window), 0, u64::from(cap)).output_cap_tokens)
370 .unwrap_or(cap)
371 .max(1)
372 }
373
374 /// Share of one output allowance a single review pass must keep for visible
375 /// text, as a percentage.
376 ///
377 /// A reasoning route shares one `max_tokens` allowance between hidden
378 /// reasoning and visible text, and Codewhale has no wire-level separation
379 /// (`thinking.budget_tokens` is not plumbed), so "reserving" means two things
380 /// together: state the reserve, and cap the reasoning level that may consume
381 /// it (`review::bounded_review_reasoning_effort`).
382 ///
383 /// The share is sized from the model's reasoning behaviour rather than a flat
384 /// constant:
385 ///
386 /// * `Some(false)` — nothing to reserve; the whole allowance is visible text.
387 /// * `Some(true)` in the summarized-reasoning families whose reasoning is
388 /// counted as ordinary output tokens
389 /// ([`codewhale_models::model_is_openai_reasoning_family`]) — half. These are
390 /// the models observed consuming an entire 64K allowance on reasoning and
391 /// returning zero visible text with stop reason `length` (#6285).
392 /// * `Some(true)` otherwise, and `None` (no catalogue row) — a quarter. A
393 /// review pass needs only enough text for its structured findings, and an
394 /// unknown model is not evidence that it does not reason (#6032).
395 #[must_use]
396 pub(crate) fn review_visible_text_reserve_percent(model: &str) -> u32 {
397 review_reserve_percent_for(
398 codewhale_models::model_reasoning_capability(model),
399 codewhale_models::model_is_openai_reasoning_family(model),
400 )
401 }
402
403 /// Pure core of [`review_visible_text_reserve_percent`]: the mapping from a
404 /// model's reasoning classification to the reserved share. Split out so the
405 /// mapping is testable without the process-global model catalog.
406 fn review_reserve_percent_for(capability: Option<bool>, openai_reasoning_family: bool) -> u32 {
407 match capability {
408 // Nothing to reserve; the whole allowance is visible text.
409 Some(false) => 0,
410 Some(true) if openai_reasoning_family => 50,
411 Some(true) | None => 25,
412 }
413 }
414
415 /// Visible-text reserve in tokens for one review pass on this exact model and
416 /// resolved output allowance.
417 #[must_use]
418 pub(crate) fn review_visible_text_reserve_tokens(model: &str, allowance: u32) -> u32 {
419 allowance.saturating_mul(review_visible_text_reserve_percent(model)) / 100
420 }
421
422 /// Output reservation used by the internal input budget for a route.
423 #[must_use]
424 pub(crate) fn route_output_reservation(
425 provider: ProviderKind,
426 model: &str,
427 route_limits: Option<RouteLimits>,
428 ) -> u32 {
429 // Use exactly the value that can reach the wire on every window size.
430 // The previous split reserved 65K for a possible 325K wire request below
431 // 500K, then jumped to an unrequested 262K reservation at 500K. Both
432 // directions made preflight disagree with the actual request. Reasoning
433 // effort is a request control, not separately metered non-wire output, so
434 // it does not justify a second hidden context reservation.
435 effective_max_output_tokens_for_route(provider, model, route_limits)
436 }
437
438 #[must_use]
439 pub(crate) fn route_context_budget(
440 provider: ProviderKind,
441 model: &str,
442 route_limits: Option<RouteLimits>,
443 input_tokens: usize,
444 ) -> Option<ContextBudget> {
445 let window = route_context_window_tokens(provider, model, route_limits);
446 let output_cap = route_output_reservation(provider, model, route_limits);
447 Some(ContextBudget::new_with_input_limit(
448 u64::from(window),
449 u64::try_from(input_tokens).ok()?,
450 u64::from(output_cap),
451 route_input_limit_tokens(route_limits).map(u64::from),
452 ))
453 }
454
455 #[must_use]
456 pub(crate) fn compaction_threshold_for_route_at_percent(
457 provider: ProviderKind,
458 model: &str,
459 route_limits: Option<RouteLimits>,
460 percent: f64,
461 ) -> usize {
462 route_context_budget(provider, model, route_limits, 0)
463 .and_then(|budget| {
464 usize::try_from(budget.compaction_trigger_for_percent(percent.clamp(10.0, 100.0))).ok()
465 })
466 .unwrap_or(DEFAULT_COMPACTION_TOKEN_THRESHOLD)
467 }
468
469 #[must_use]
470 pub(crate) fn auto_compact_default_for_route(
471 provider: ProviderKind,
472 model: &str,
473 route_limits: Option<RouteLimits>,
474 ) -> bool {
475 // Every resolved route has either concrete offering limits or a
476 // conservative provider/model fallback. Large windows need continuity too;
477 // their size is not a reason to disable compaction entirely.
478 route_context_window_tokens(provider, model, route_limits) > 0
479 }
480
481 #[cfg(test)]
482 mod tests {
483 #[test]
484 fn route_inline_char_budget_is_three_percent_of_the_window_capped() {
485 use super::route_inline_char_budget_with_raise as budget;
486 assert_eq!(budget(Some(128_000), None), 15_360);
487 assert_eq!(budget(Some(10_000), None), 1_200);
488 assert_eq!(budget(None, None), 100_000);
489 assert_eq!(budget(Some(0), None), 100_000);
490 assert_eq!(budget(Some(1_000_000), None), 100_000);
491 // An operator opt-in raises the budget, never lowers it, and stops at 2 MiB.
492 assert_eq!(budget(Some(128_000), Some(80_000)), 80_000);
493 assert_eq!(budget(Some(128_000), Some(1_000)), 15_360);
494 assert_eq!(
495 budget(Some(128_000), Some(64 * 1024 * 1024)),
496 2 * 1024 * 1024
497 );
498 }
499
500 use super::*;
501
502 #[test]
503 fn provider_regression_5820_small_unknown_windows_keep_room_for_input() {
504 let _lock = crate::test_support::lock_test_env();
505 let _canonical = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
506 let _legacy = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
507 let model = "qwen2.5:7b";
508 assert!(codewhale_models::max_output_tokens_for_model(model).is_none());
509 for provider in [
510 ProviderKind::Ollama,
511 ProviderKind::Sglang,
512 ProviderKind::Vllm,
513 ProviderKind::Custom,
514 ] {
515 for (window, expected_output) in [
516 (16_384, 4_096),
517 (32_768, 8_192),
518 (65_536, 16_384),
519 (262_144, 65_536),
520 ] {
521 let limits = Some(RouteLimits {
522 context_tokens: Some(window),
523 ..RouteLimits::default()
524 });
525 let wire_cap = effective_max_output_tokens_for_route(provider, model, limits);
526 let budget = route_context_budget(provider, model, limits, 6_225).unwrap();
527 assert_eq!(wire_cap, expected_output, "{provider:?}, window={window}");
528 assert_eq!(budget.output_cap_tokens, u64::from(wire_cap));
529 assert_eq!(
530 budget.input_budget_ceiling,
531 window - u64::from(wire_cap) - 1_024
532 );
533 assert!(
534 budget.input_tokens < budget.input_budget_ceiling,
535 "{budget:?}"
536 );
537 assert!(!budget.should_compact(), "{budget:?}");
538 }
539 }
540 let _explicit =
541 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", "16384");
542 let limits = RouteLimits {
543 context_tokens: Some(32_768),
544 ..RouteLimits::default()
545 };
546 assert_eq!(
547 effective_max_output_tokens_for_route(ProviderKind::Ollama, model, Some(limits)),
548 16_384
549 );
550 assert_eq!(
551 effective_max_output_tokens_for_route(
552 ProviderKind::Ollama,
553 model,
554 Some(RouteLimits {
555 output_tokens: Some(4_096),
556 ..limits
557 })
558 ),
559 4_096
560 );
561 }
562
563 /// #6540: the runtime store's 15 failed compactions were all emergency
564 /// passes on an unknown local Ollama tag (`qwen3:4b`) with no route
565 /// limits: the capability fallback window (8K) minus a 6K output
566 /// reservation left a ~1K input budget, so every first request of a turn
567 /// tripped preflight recovery. The fallback window must keep the same
568 /// input room a configured window of that size gets.
569 #[test]
570 fn provider_regression_6540_fallback_window_keeps_room_for_input() {
571 let _lock = crate::test_support::lock_test_env();
572 let _canonical = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
573 let _legacy = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
574 let model = "qwen3:4b";
575 assert!(codewhale_models::max_output_tokens_for_model(model).is_none());
576 let window = route_context_window_tokens(ProviderKind::Ollama, model, None);
577 assert_eq!(
578 window, 8_192,
579 "unknown local tags keep the conservative window"
580 );
581
582 let wire_cap = effective_max_output_tokens_for_route(ProviderKind::Ollama, model, None);
583 assert_eq!(wire_cap, 2_048);
584 let budget = route_context_budget(ProviderKind::Ollama, model, None, 0).unwrap();
585 assert_eq!(budget.output_cap_tokens, u64::from(wire_cap));
586 assert_eq!(budget.input_budget_ceiling, 8_192 - 2_048 - 1_024);
587 // The recorded first-request estimates (~1.9K–3.9K) now fit.
588 assert!(budget.input_budget_ceiling > 3_900, "{budget:?}");
589
590 // Same answer as the explicitly configured 8K window (#5820).
591 let configured = Some(RouteLimits {
592 context_tokens: Some(8_192),
593 ..RouteLimits::default()
594 });
595 assert_eq!(
596 effective_max_output_tokens_for_route(ProviderKind::Ollama, model, configured),
597 wire_cap
598 );
599 }
600
601 /// Absence of a catalogue row is not evidence of a large ceiling. An
602 /// unrecognized wire alias on a remote OpenAI-compatible route keeps the
603 /// conservative compatibility ceiling, with an attributable source.
604 #[test]
605 fn uncatalogued_remote_model_keeps_a_conservative_ceiling() {
606 let _lock = crate::test_support::lock_test_env();
607 let _canonical = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
608 let _legacy = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
609 let source = output_ceiling_source(ProviderKind::Openai, "totally-unknown-alias-v9");
610 assert_eq!(
611 source,
612 OutputCeilingSource::Uncatalogued(UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS)
613 );
614 assert_eq!(
615 source.clamp_tokens(),
616 Some(UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS)
617 );
618 assert!(
619 effective_max_output_tokens_for_route(
620 ProviderKind::Openai,
621 "totally-unknown-alias-v9",
622 None
623 ) <= UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS
624 );
625 }
626
627 /// #5460: absence is not permission, but a positive output maximum on the
628 /// resolved route is permission for that exact route. The concrete fact
629 /// replaces only the catalogue-absence guess; it never raises the caller's
630 /// requested cap or a documented model ceiling.
631 #[test]
632 fn concrete_route_output_limit_outranks_uncatalogued_guess() {
633 let _lock = crate::test_support::lock_test_env();
634 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
635 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
636 let model = "totally-unknown-alias-v9";
637
638 assert_eq!(effective_max_output_tokens(model), 64_000);
639 for provider in [ProviderKind::Openai, ProviderKind::Custom] {
640 assert_eq!(
641 output_ceiling_source(provider, model),
642 OutputCeilingSource::Uncatalogued(UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS)
643 );
644 assert_eq!(
645 effective_max_output_tokens_for_route(provider, model, None),
646 64_000,
647 "{provider:?}: no route fact must preserve the labeled automatic allowance"
648 );
649 for route_cap in [24_576, 64_000] {
650 assert_eq!(
651 effective_max_output_tokens_for_route(
652 provider,
653 model,
654 Some(RouteLimits {
655 output_tokens: Some(route_cap),
656 ..RouteLimits::default()
657 }),
658 ),
659 u32::try_from(route_cap).unwrap(),
660 "{provider:?}: the exact route fact must replace the catalogue-absence guess"
661 );
662 }
663 assert_eq!(
664 effective_max_output_tokens_for_route(
665 provider,
666 model,
667 Some(RouteLimits {
668 output_tokens: Some(65_536),
669 ..RouteLimits::default()
670 }),
671 ),
672 64_000,
673 "{provider:?}: a route fact must not raise the requested cap"
674 );
675 }
676
677 assert_eq!(
678 effective_max_output_tokens_for_route(
679 ProviderKind::Moonshot,
680 "kimi-k2.7-code",
681 Some(RouteLimits {
682 output_tokens: Some(64_000),
683 ..RouteLimits::default()
684 }),
685 ),
686 32_768,
687 "a route fact must not raise a documented model ceiling"
688 );
689 }
690
691 /// Routes that *declare* an unknown ceiling still avoid the clamp.
692 #[test]
693 fn route_declared_unknown_ceilings_are_not_clamped() {
694 for (provider, model) in [
695 (ProviderKind::Moonshot, "kimi-for-coding"),
696 (ProviderKind::Moonshot, "kimi-for-coding-highspeed"),
697 (ProviderKind::Ollama, "some-local-build"),
698 ] {
699 assert_eq!(
700 output_ceiling_source(provider, model),
701 OutputCeilingSource::RouteDeclaredUnknown,
702 "{provider:?}/{model} must declare its unknown ceiling"
703 );
704 assert_eq!(output_ceiling_source(provider, model).clamp_tokens(), None);
705 }
706 // Bare `k3` is a membership id, but unlike the `kimi-for-coding`
707 // family the K3 quickstart documents its output maximum, and the model
708 // catalogue carries it. A documented ceiling is authoritative — the
709 // membership allowlist only covers ids the catalogue has nothing to
710 // say about, and must not turn a real fact back into an unknown.
711 assert_eq!(
712 output_ceiling_source(ProviderKind::Moonshot, "k3"),
713 OutputCeilingSource::Documented(131_072)
714 );
715 assert_eq!(
716 output_ceiling_source(ProviderKind::OllamaCloud, "some-cloud-build"),
717 OutputCeilingSource::Uncatalogued(UNCATALOGUED_COMPAT_MAX_OUTPUT_TOKENS),
718 "hosted Ollama Cloud must not inherit the local runtime's unbounded output semantics"
719 );
720 }
721
722 /// #5440: an Anthropic-family model the catalogue does not describe keeps
723 /// the 64K Messages floor as its clamp, but the floor is an assumption
724 /// about that model — never a "documented" ceiling.
725 #[test]
726 fn anthropic_unknown_model_ceiling_is_an_unverified_assumed_floor() {
727 let source = output_ceiling_source(ProviderKind::Anthropic, "claude-future-99");
728 assert_eq!(source, OutputCeilingSource::Unverified(64_000));
729 assert_eq!(source.as_str(), "unverified");
730 assert_eq!(source.clamp_tokens(), Some(64_000));
731 // Same honesty on the compatibility dialects of the same family.
732 assert_eq!(
733 output_ceiling_source(ProviderKind::MinimaxAnthropic, "claude-future-99"),
734 OutputCeilingSource::Unverified(64_000)
735 );
736 assert_eq!(
737 output_ceiling_source(ProviderKind::Openmodel, "claude-future-99"),
738 OutputCeilingSource::Unverified(64_000)
739 );
740 }
741
742 /// #5440: a model the catalogue does describe keeps its documented
743 /// ceiling and its documented label — the unverified rung must not
744 /// swallow real facts.
745 #[test]
746 fn anthropic_documented_model_ceiling_stays_documented() {
747 let source = output_ceiling_source(ProviderKind::Anthropic, "claude-sonnet-4-6");
748 assert_eq!(source, OutputCeilingSource::Documented(128_000));
749 assert_eq!(source.as_str(), "documented");
750 assert_eq!(source.clamp_tokens(), Some(128_000));
751 }
752
753 /// #5440: the Codex OAuth route clamps every response to 4K by policy,
754 /// because the OAuth cache publishes no ceiling. The clamp stands; the
755 /// receipt must call the number what it is.
756 #[test]
757 fn codex_oauth_ceiling_clamps_but_never_claims_documented() {
758 let source = output_ceiling_source(ProviderKind::OpenaiCodex, "gpt-5.5");
759 assert_eq!(source, OutputCeilingSource::Unverified(4_096));
760 assert_eq!(source.as_str(), "unverified");
761 assert_eq!(source.clamp_tokens(), Some(4_096));
762 assert_eq!(
763 effective_max_output_tokens_for_route(ProviderKind::OpenaiCodex, "gpt-5.5", None),
764 4_096,
765 "the honesty relabel must not revalue the long-standing clamp"
766 );
767 }
768
769 #[test]
770 fn codex_missing_route_metadata_uses_provider_context_floor() {
771 assert_eq!(
772 route_context_window_tokens(ProviderKind::OpenaiCodex, "gpt-5.5", None),
773 128_000
774 );
775 // 80% of the 128K window (102_400) fits under the input ceiling.
776 assert_eq!(
777 compaction_threshold_for_route_at_percent(
778 ProviderKind::OpenaiCodex,
779 "gpt-5.5",
780 None,
781 80.0,
782 ),
783 102_400
784 );
785 assert!(auto_compact_default_for_route(
786 ProviderKind::OpenaiCodex,
787 "gpt-5.5",
788 None,
789 ));
790 }
791
792 /// The assertion values here depend on `explicit_max_output_tokens_override`
793 /// seeing no ambient env override, and sibling tests in this binary
794 /// (this module, `client`, `vision/tools`, `core/engine`) set
795 /// `CODEWHALE_MAX_OUTPUT_TOKENS`/`DEEPSEEK_MAX_OUTPUT_TOKENS` while holding
796 /// `lock_test_env`. Without the lock and guards this test could read a
797 /// concurrent writer's value mid-assertion (process-global env, parallel
798 /// threads), which is the order-dependent flake this guards against.
799 #[test]
800 fn v4_trigger_uses_window_percent_when_it_fits_spendable_input() {
801 let _lock = crate::test_support::lock_test_env();
802 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
803 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
804
805 let budget = route_context_budget(ProviderKind::Deepseek, "deepseek-v4-pro", None, 0)
806 .expect("V4 route budget");
807
808 assert_eq!(budget.window_tokens, 1_000_000);
809 assert_eq!(budget.output_cap_tokens, u64::from(API_MAX_OUTPUT_TOKENS));
810 assert_eq!(budget.input_budget_ceiling, 933_440);
811 // 80% of the 1M window fits below the spendable input ceiling.
812 assert_eq!(
813 compaction_threshold_for_route_at_percent(
814 ProviderKind::Deepseek,
815 "deepseek-v4-pro",
816 None,
817 80.0,
818 ),
819 800_000
820 );
821 }
822
823 #[test]
824 fn kimi_k3_defaults_auto_compaction_on() {
825 assert!(auto_compact_default_for_route(
826 ProviderKind::Moonshot,
827 "kimi-k3",
828 None,
829 ));
830 }
831
832 #[test]
833 fn kimi_catalog_output_ceiling_preserves_input_budget() {
834 let _lock = crate::test_support::lock_test_env();
835 let _max_output = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
836 // #4368/#4378: Models.dev may report Kimi's full 262K context as both
837 // context and output ceilings. Reserve the route-effective 32K request
838 // cap rather than treating that catalog maximum as the amount every
839 // turn will emit.
840 let limits = RouteLimits {
841 context_tokens: Some(262_144),
842 output_tokens: Some(262_144),
843 ..RouteLimits::default()
844 };
845 let budget =
846 route_context_budget(ProviderKind::Moonshot, "kimi-k2.7-code", Some(limits), 0)
847 .expect("Kimi route budget");
848 let trigger = compaction_threshold_for_route_at_percent(
849 ProviderKind::Moonshot,
850 "kimi-k2.7-code",
851 Some(limits),
852 80.0,
853 );
854
855 assert_eq!(budget.output_cap_tokens, 32_768);
856 assert_eq!(budget.input_budget_ceiling, 228_352);
857 // 80% of the 262_144 window; fits under the 228_352 ceiling because
858 // the output reservation is the route-effective 32K request cap.
859 assert_eq!(trigger, 209_715);
860 assert!(trigger as u64 <= budget.input_budget_ceiling);
861 }
862
863 #[test]
864 fn explicit_route_output_limit_beats_unknown_model_name_fallback() {
865 let _lock = crate::test_support::lock_test_env();
866 let _max_output =
867 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", "65536");
868 let limits = RouteLimits {
869 context_tokens: Some(262_144),
870 output_tokens: Some(24_576),
871 ..RouteLimits::default()
872 };
873
874 assert_eq!(
875 effective_max_output_tokens_for_route(
876 ProviderKind::Vllm,
877 "arbitrary-local-wire-alias",
878 Some(limits),
879 ),
880 24_576
881 );
882 assert_eq!(
883 effective_max_output_tokens_for_route(
884 ProviderKind::Vllm,
885 "arbitrary-local-wire-alias",
886 None,
887 ),
888 65_536,
889 "an unknown compatibility cap must not clamp; only the requested cap applies"
890 );
891 assert_eq!(
892 effective_max_output_tokens_for_route(
893 ProviderKind::Vllm,
894 "kimi-k2.7-code",
895 Some(RouteLimits {
896 output_tokens: Some(262_144),
897 ..RouteLimits::default()
898 }),
899 ),
900 32_768,
901 "known model caps must remain authoritative on self-hosted routes"
902 );
903 }
904
905 /// #4368 follow-up: the Kimi Code membership ids deliberately have no
906 /// static output cap (the membership catalog owns their limits). The old
907 /// generic `unwrap_or(4096)` in `provider_capability` turned that unknown
908 /// into a hard 4K clamp here, silently truncating every offline membership
909 /// turn. Unknown must mean "no compatibility clamp".
910 #[test]
911 fn kimi_membership_unknown_output_cap_does_not_clamp_to_4k() {
912 let _lock = crate::test_support::lock_test_env();
913 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
914 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
915
916 for model in ["kimi-for-coding", "kimi-for-coding-highspeed"] {
917 assert_eq!(
918 provider_capability(ProviderKind::Moonshot, model).max_output,
919 None,
920 "{model}: membership output ceiling must stay unknown, not a placeholder"
921 );
922
923 let cap = effective_max_output_tokens_for_route(ProviderKind::Moonshot, model, None);
924 assert_eq!(
925 cap,
926 effective_max_output_tokens(model),
927 "{model}: unknown compatibility cap must leave the requested cap intact"
928 );
929 assert_ne!(cap, 4_096, "{model}: must not inherit the old 4K fallback");
930 // No invented sentinel ceiling either.
931 assert_ne!(cap, u32::MAX);
932 assert_ne!(cap, 32_768);
933 }
934 }
935
936 /// A concrete membership offering limit is still authoritative — "unknown
937 /// means no clamp" must not become "never clamp".
938 #[test]
939 fn kimi_membership_route_limit_still_caps_output() {
940 let _lock = crate::test_support::lock_test_env();
941 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
942 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
943
944 let limits = RouteLimits {
945 context_tokens: Some(262_144),
946 output_tokens: Some(16_384),
947 ..RouteLimits::default()
948 };
949 assert_eq!(
950 effective_max_output_tokens_for_route(
951 ProviderKind::Moonshot,
952 "kimi-for-coding",
953 Some(limits),
954 ),
955 16_384
956 );
957 }
958
959 /// GLM and MiniMax publish real output ceilings; those stay authoritative
960 /// so relaxing the unknown case cannot leak into known routes.
961 #[test]
962 fn known_glm_and_minimax_output_caps_remain_authoritative() {
963 let _lock = crate::test_support::lock_test_env();
964 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
965 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
966
967 // GLM 5.2: 1M window, documented 131K output. The capability remains
968 // known even though the safe automatic request starts at 64K.
969 let glm = provider_capability(ProviderKind::Zai, "glm-5.2");
970 assert_eq!(glm.max_output, Some(131_072));
971
972 let minimax = provider_capability(ProviderKind::Minimax, "minimax-m3");
973 assert_eq!(minimax.max_output, Some(524_288));
974
975 // A known cap below the requested cap must still clamp.
976 assert_eq!(
977 effective_max_output_tokens_for_route(ProviderKind::Moonshot, "kimi-k2.7-code", None),
978 32_768,
979 );
980 }
981
982 /// A documented capability maximum is not itself a sane default request
983 /// size. The capability remains documented and available as an explicit
984 /// override; only the automatic request is bounded.
985 #[test]
986 fn documented_ceiling_is_a_bound_not_an_unbounded_default_request() {
987 let _lock = crate::test_support::lock_test_env();
988 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
989 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
990
991 for model in [
992 "deepseek-v4-flash",
993 "deepseek-v4-pro",
994 "deepseek-v4flash",
995 "deepseek-ai/deepseek-v4-pro",
996 "deepseek-chat",
997 "deepseek-reasoner",
998 ] {
999 assert_eq!(
1000 output_ceiling_source(ProviderKind::Deepseek, model),
1001 OutputCeilingSource::Documented(384_000),
1002 "{model}"
1003 );
1004 }
1005 assert_eq!(
1006 effective_max_output_tokens("deepseek-v4-flash"),
1007 API_MAX_OUTPUT_TOKENS,
1008 "a 384K capability maximum must not become the no-config request size"
1009 );
1010 assert_eq!(
1011 effective_max_output_tokens("glm-5.2"),
1012 API_MAX_OUTPUT_TOKENS,
1013 "a 131K capability maximum must also remain a ceiling, not a default"
1014 );
1015 }
1016
1017 #[test]
1018 fn uncatalogued_deepseek_variants_require_exact_output_metadata() {
1019 let _env_lock = crate::test_support::lock_test_env();
1020 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
1021 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1022
1023 for provider in [
1024 ProviderKind::Deepseek,
1025 ProviderKind::Deepseek,
1026 ProviderKind::DeepseekAnthropic,
1027 ProviderKind::Custom,
1028 ] {
1029 for model in [
1030 "deepseek-v4.1-flash-expires-on-0910",
1031 "deepseek-v4.1-flash",
1032 "deepseek-v4-flash-vendor",
1033 ] {
1034 assert_eq!(provider_capability(provider, model).max_output, None);
1035 let source = output_ceiling_source(provider, model);
1036 assert_eq!(source, OutputCeilingSource::Uncatalogued(65_536));
1037 assert_eq!(source.as_str(), "uncatalogued");
1038 assert_eq!(
1039 effective_max_output_tokens_for_route(provider, model, None),
1040 64_000,
1041 "{provider:?}: {model}"
1042 );
1043 }
1044 }
1045
1046 // Exact route output facts supply a ceiling only for that request.
1047 let limits = RouteLimits {
1048 context_tokens: Some(128_000),
1049 output_tokens: Some(24_576),
1050 ..RouteLimits::default()
1051 };
1052 assert_eq!(
1053 effective_max_output_tokens_for_route(
1054 ProviderKind::Deepseek,
1055 "deepseek-v4.1-flash-expires-on-0910",
1056 Some(limits)
1057 ),
1058 24_576
1059 );
1060 assert_eq!(
1061 codewhale_models::max_output_tokens_for_model("deepseek-v4.1-flash-expires-on-0910"),
1062 None
1063 );
1064 }
1065
1066 #[test]
1067 fn deepseek_v4_explicit_mid_windows_share_one_safe_no_config_budget() {
1068 let _lock = crate::test_support::lock_test_env();
1069 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
1070 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1071
1072 for window in [262_144, 327_680, 393_216] {
1073 let limits = RouteLimits {
1074 context_tokens: Some(window),
1075 ..RouteLimits::default()
1076 };
1077 let cap = effective_max_output_tokens_for_route(
1078 ProviderKind::Vllm,
1079 "DeepSeek-V4-Flash",
1080 Some(limits),
1081 );
1082 let reservation =
1083 route_output_reservation(ProviderKind::Vllm, "DeepSeek-V4-Flash", Some(limits));
1084 let budget = route_context_budget(
1085 ProviderKind::Vllm,
1086 "DeepSeek-V4-Flash",
1087 Some(limits),
1088 105_000,
1089 )
1090 .expect("explicit vLLM route budget");
1091
1092 assert_eq!(cap, API_MAX_OUTPUT_TOKENS, "window={window}");
1093 assert_eq!(reservation, cap, "window={window}");
1094 assert_eq!(
1095 budget.input_budget_ceiling,
1096 window - u64::from(API_MAX_OUTPUT_TOKENS) - 1_024,
1097 "window={window}"
1098 );
1099 assert!(
1100 105_000 < budget.input_budget_ceiling,
1101 "ordinary 85K-105K inputs must not trigger emergency compaction: {budget:?}"
1102 );
1103 assert!(budget.available_input_tokens > 0, "window={window}");
1104 }
1105 }
1106
1107 #[test]
1108 fn explicit_output_override_is_preserved_and_reserved_on_mid_windows() {
1109 let _lock = crate::test_support::lock_test_env();
1110 let _codewhale =
1111 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", "100000");
1112 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1113 let limits = RouteLimits {
1114 context_tokens: Some(327_680),
1115 ..RouteLimits::default()
1116 };
1117
1118 let cap = effective_max_output_tokens_for_route(
1119 ProviderKind::Vllm,
1120 "DeepSeek-V4-Flash",
1121 Some(limits),
1122 );
1123 assert_eq!(cap, 100_000);
1124 assert_eq!(
1125 route_output_reservation(ProviderKind::Vllm, "DeepSeek-V4-Flash", Some(limits),),
1126 cap
1127 );
1128 let budget = route_context_budget(
1129 ProviderKind::Vllm,
1130 "DeepSeek-V4-Flash",
1131 Some(limits),
1132 105_000,
1133 )
1134 .expect("override route budget");
1135 assert_eq!(budget.input_budget_ceiling, 226_656);
1136 assert!(budget.available_input_tokens > 0);
1137 }
1138
1139 #[test]
1140 fn explicit_uncatalogued_allowance_respects_route_and_context_limits() {
1141 let _lock = crate::test_support::lock_test_env();
1142 let _canonical =
1143 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", "100000");
1144 let _legacy = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1145 let model = "uncatalogued-preview-for-output-test";
1146 let limits = RouteLimits {
1147 context_tokens: Some(327_680),
1148 ..RouteLimits::default()
1149 };
1150 assert_eq!(
1151 effective_max_output_tokens_for_route(ProviderKind::Custom, model, Some(limits)),
1152 100_000
1153 );
1154 assert_eq!(
1155 effective_max_output_tokens_for_route(
1156 ProviderKind::Custom,
1157 model,
1158 Some(RouteLimits {
1159 output_tokens: Some(32_768),
1160 ..limits
1161 })
1162 ),
1163 32_768
1164 );
1165 let small = RouteLimits {
1166 context_tokens: Some(32_768),
1167 ..RouteLimits::default()
1168 };
1169 let cap = effective_max_output_tokens_for_route(ProviderKind::Custom, model, Some(small));
1170 assert_eq!(cap, 30_720);
1171 assert_eq!(
1172 route_output_reservation(ProviderKind::Custom, model, Some(small)),
1173 cap
1174 );
1175 }
1176
1177 #[test]
1178 fn oversized_explicit_override_is_clamped_and_reserved_to_the_route_window() {
1179 let _lock = crate::test_support::lock_test_env();
1180 let _codewhale =
1181 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", "384000");
1182 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1183 let limits = RouteLimits {
1184 context_tokens: Some(327_680),
1185 ..RouteLimits::default()
1186 };
1187
1188 let cap = effective_max_output_tokens_for_route(
1189 ProviderKind::Vllm,
1190 "DeepSeek-V4-Flash",
1191 Some(limits),
1192 );
1193 assert_eq!(cap, 325_632);
1194 assert_eq!(
1195 route_output_reservation(ProviderKind::Vllm, "DeepSeek-V4-Flash", Some(limits),),
1196 cap,
1197 "preflight must reserve every token the explicit override can put on the wire"
1198 );
1199 let budget = route_context_budget(ProviderKind::Vllm, "DeepSeek-V4-Flash", Some(limits), 0)
1200 .expect("oversized override route budget");
1201 assert_eq!(budget.input_budget_ceiling, 1_024);
1202 }
1203
1204 #[test]
1205 fn explicit_override_on_large_window_stays_unified() {
1206 let _lock = crate::test_support::lock_test_env();
1207 let _codewhale =
1208 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", "384000");
1209 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1210 let limits = RouteLimits {
1211 context_tokens: Some(1_000_000),
1212 ..RouteLimits::default()
1213 };
1214
1215 let cap = effective_max_output_tokens_for_route(
1216 ProviderKind::Vllm,
1217 "DeepSeek-V4-Flash",
1218 Some(limits),
1219 );
1220 let reservation =
1221 route_output_reservation(ProviderKind::Vllm, "DeepSeek-V4-Flash", Some(limits));
1222 let budget = route_context_budget(ProviderKind::Vllm, "DeepSeek-V4-Flash", Some(limits), 0)
1223 .expect("large explicit route budget");
1224
1225 assert_eq!(cap, 384_000);
1226 assert_eq!(reservation, cap);
1227 assert_eq!(budget.output_cap_tokens, u64::from(cap));
1228 assert_eq!(budget.input_budget_ceiling, 614_976);
1229 }
1230
1231 #[test]
1232 fn automatic_wire_cap_and_reservation_have_no_large_window_cliff() {
1233 let _lock = crate::test_support::lock_test_env();
1234 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
1235 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1236
1237 for window in [499_999, 500_000, 1_000_000] {
1238 let limits = RouteLimits {
1239 context_tokens: Some(window),
1240 ..RouteLimits::default()
1241 };
1242 let wire = effective_max_output_tokens_for_route(
1243 ProviderKind::Vllm,
1244 "DeepSeek-V4-Flash",
1245 Some(limits),
1246 );
1247 let reservation =
1248 route_output_reservation(ProviderKind::Vllm, "DeepSeek-V4-Flash", Some(limits));
1249 assert_eq!(wire, API_MAX_OUTPUT_TOKENS, "window={window}");
1250 assert_eq!(reservation, wire, "window={window}");
1251 }
1252 }
1253
1254 #[test]
1255 fn concrete_route_input_limit_clamps_preflight_and_compaction() {
1256 let _lock = crate::test_support::lock_test_env();
1257 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
1258 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1259 let limits = RouteLimits {
1260 context_tokens: Some(1_000_000),
1261 input_tokens: Some(128_000),
1262 output_tokens: Some(64_000),
1263 };
1264
1265 let budget = route_context_budget(
1266 ProviderKind::Vllm,
1267 "DeepSeek-V4-Flash",
1268 Some(limits),
1269 200_000,
1270 )
1271 .expect("route budget");
1272 assert_eq!(route_input_limit_tokens(Some(limits)), Some(128_000));
1273 assert_eq!(budget.input_budget_ceiling, 128_000);
1274 assert_eq!(budget.available_input_tokens, 0);
1275 assert_eq!(budget.compaction_trigger_for_percent(80.0), 128_000);
1276 }
1277
1278 #[test]
1279 fn canonical_output_override_blank_falls_through_but_invalid_is_authoritative() {
1280 let _lock = crate::test_support::lock_test_env();
1281 let _legacy = crate::test_support::EnvVarGuard::set("DEEPSEEK_MAX_OUTPUT_TOKENS", "100000");
1282
1283 {
1284 let _canonical =
1285 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", " ");
1286 assert_eq!(explicit_max_output_tokens_override(), Some(100_000));
1287 }
1288 for invalid in ["not-a-number", "0"] {
1289 let _canonical =
1290 crate::test_support::EnvVarGuard::set("CODEWHALE_MAX_OUTPUT_TOKENS", invalid);
1291 assert_eq!(explicit_max_output_tokens_override(), None, "{invalid}");
1292 assert_eq!(
1293 effective_max_output_tokens("deepseek-v4-pro"),
1294 API_MAX_OUTPUT_TOKENS
1295 );
1296 }
1297 }
1298
1299 #[test]
1300 fn mid_window_internal_reservation_stays_on_the_ordinary_request_floor() {
1301 let _lock = crate::test_support::lock_test_env();
1302 let _codewhale = crate::test_support::EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
1303 let _deepseek = crate::test_support::EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1304 let reservation =
1305 route_output_reservation(ProviderKind::Arcee, "trinity-large-thinking", None);
1306 assert_eq!(reservation, API_MAX_OUTPUT_TOKENS);
1307 let budget = route_context_budget(ProviderKind::Arcee, "trinity-large-thinking", None, 0)
1308 .expect("trinity route budget");
1309 assert_eq!(budget.compaction_trigger_for_percent(80.0), 195_584);
1310 }
1311
1312 #[test]
1313 fn review_reserve_is_sized_from_reasoning_classification() {
1314 // Mapping core (#6285): every classification arm.
1315 assert_eq!(review_reserve_percent_for(Some(false), false), 0);
1316 assert_eq!(review_reserve_percent_for(Some(false), true), 0);
1317 assert_eq!(review_reserve_percent_for(Some(true), false), 25);
1318 assert_eq!(review_reserve_percent_for(Some(true), true), 50);
1319 // Unknown (no catalogue row) is not evidence of no reasoning (#6032).
1320 assert_eq!(review_reserve_percent_for(None, false), 25);
1321 assert_eq!(review_reserve_percent_for(None, true), 25);
1322 }
1323
1324 #[test]
1325 fn review_reserve_tokens_follow_the_model_classification() {
1326 // A model no catalogue row resolves for: a quarter of the allowance is
1327 // reserved as visible text, and the token math scales off the exact
1328 // resolved allowance.
1329 let unknown = "not-a-catalogue-model-6285";
1330 assert_eq!(review_visible_text_reserve_percent(unknown), 25);
1331 assert_eq!(review_visible_text_reserve_tokens(unknown, 65_536), 16_384);
1332 assert_eq!(review_visible_text_reserve_tokens(unknown, 0), 0);
1333 }
1334 }
1335
1335 lines RUST