返回 CodeWhale
test_cases_14.rs
根目录 / crates / tui / src / core / engine / tests / test_cases_14.rs
1 #[tokio::test]
2 async fn sync_session_projects_persisted_subagent_handoff_for_headless_restore() {
3 let tmp = tempdir().expect("tempdir");
4 let config = EngineConfig {
5 workspace: tmp.path().to_path_buf(),
6 model: "deepseek-v4-pro".to_string(),
7 ..Default::default()
8 };
9 let (engine, handle) = Engine::new(config, &Config::default());
10 let payload = concat!(
11 "Child result retained.\nCheckpoint: engine restore is covered.\n",
12 "<codewhale:subagent.done>{\"agent_id\":\"agent_headless\",",
13 "\"status\":\"completed\",\"summary_location\":\"previous_line\"}",
14 "</codewhale:subagent.done>",
15 );
16 let messages = vec![
17 Message {
18 role: Role::User,
19 content: vec![ContentBlock::Text {
20 text: "Keep the original task".to_string(),
21 cache_control: None,
22 }],
23 },
24 crate::runtime_handoff::subagent_completion_runtime_message(payload),
25 ];
26
27 let run = tokio::spawn(engine.run());
28 handle
29 .send(Op::SyncSession {
30 session_id: Some("headless-resume".to_string()),
31 messages,
32 system_prompt: None,
33 system_prompt_override: false,
34 model: "deepseek-v4-pro".to_string(),
35 workspace: tmp.path().to_path_buf(),
36 mode: AppMode::Agent,
37 })
38 .await
39 .expect("sync session");
40
41 let (tx, rx) = tokio::sync::oneshot::channel();
42 handle
43 .send(Op::GetSessionSnapshot {
44 tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))),
45 })
46 .await
47 .expect("request snapshot");
48 let snapshot = tokio::time::timeout(Duration::from_secs(2), rx)
49 .await
50 .expect("snapshot response")
51 .expect("snapshot");
52
53 assert_eq!(snapshot.messages.len(), 2);
54 assert!(snapshot.messages[0].content.iter().any(
55 |block| matches!(block, ContentBlock::Text { text, .. } if text == "Keep the original task")
56 ));
57 let restored =
58 crate::runtime_handoff::restored_subagent_checkpoint_display(&snapshot.messages[1])
59 .expect("projected headless checkpoint");
60 assert!(restored.contains("agent_headless"));
61 assert!(restored.contains("Checkpoint: engine restore is covered."));
62 assert!(!restored.contains("runtime_event"));
63 assert!(!restored.contains("subagent.done"));
64
65 run.abort();
66 }
67
68 #[tokio::test]
69 async fn session_snapshot_records_the_literal_custom_table_id() {
70 let tmp = tempdir().expect("tempdir");
71 let api_config = crate::config::parse_config_base(
72 r#"provider = "custom"
73 [providers.custom]
74 kind = "openai-compatible"
75 base_url = "http://127.0.0.1:18180/v1"
76 model = "legacy-root-model"
77 auth_mode = "none"
78 "#,
79 )
80 .expect("canonical literal custom table");
81 let config = EngineConfig {
82 workspace: tmp.path().to_path_buf(),
83 model: "legacy-root-model".to_string(),
84 ..Default::default()
85 };
86 let (engine, handle) = Engine::new(config, &api_config);
87
88 let run = tokio::spawn(engine.run());
89 let (tx, rx) = tokio::sync::oneshot::channel();
90 handle
91 .send(Op::GetSessionSnapshot {
92 tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))),
93 })
94 .await
95 .expect("request snapshot");
96 let snapshot = tokio::time::timeout(Duration::from_secs(2), rx)
97 .await
98 .expect("snapshot response")
99 .expect("snapshot");
100
101 // The literal route is the `[providers.custom]` table since #6394.
102 assert_eq!(snapshot.model_provider, "custom");
103 assert_eq!(snapshot.model_provider_id.as_deref(), Some("custom"));
104 run.abort();
105 }
106
107 #[tokio::test]
108 #[allow(clippy::await_holding_lock)]
109 async fn edit_last_turn_preserves_current_mode() {
110 use wiremock::matchers::{method, path};
111 use wiremock::{Mock, MockServer, ResponseTemplate};
112
113 // EditLastTurn dispatches a real replacement turn. Pin that turn to a
114 // local, completing SSE response instead of depending on whichever
115 // provider configuration or network state the parallel test process has.
116 let _lock = lock_test_env();
117 let tmp = tempdir().expect("tempdir");
118 let server = MockServer::start().await;
119 let done_sse = concat!(
120 "data: {\"id\":\"chatcmpl-edit-mode\",\"choices\":[{\"index\":0,",
121 "\"delta\":{\"content\":\"Revised plan.\"},\"finish_reason\":null}]}\n\n",
122 "data: {\"id\":\"chatcmpl-edit-mode\",\"choices\":[{\"index\":0,",
123 "\"delta\":{},\"finish_reason\":\"stop\"}]}\n\n",
124 "data: [DONE]\n\n",
125 );
126 Mock::given(method("POST"))
127 .and(path("/v1/chat/completions"))
128 .respond_with(
129 ResponseTemplate::new(200)
130 .insert_header("content-type", "text/event-stream")
131 .set_body_string(done_sse),
132 )
133 .expect(1)
134 .mount(&server)
135 .await;
136
137 let api_config = Config {
138 ..Config::default()
139 }
140 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
141 let config = EngineConfig {
142 workspace: tmp.path().to_path_buf(),
143 model: "deepseek-v4-pro".to_string(),
144 snapshots_enabled: false,
145 subagents_enabled: false,
146 ..Default::default()
147 };
148 let (engine, handle) = Engine::new(config, &api_config);
149
150 let run = tokio::spawn(engine.run());
151 let seeded_messages = vec![
152 Message {
153 role: Role::User,
154 content: vec![ContentBlock::Text {
155 text: "draft the plan".to_string(),
156 cache_control: None,
157 }],
158 },
159 Message {
160 role: Role::Assistant,
161 content: vec![ContentBlock::Text {
162 text: "initial response".to_string(),
163 cache_control: None,
164 }],
165 },
166 ];
167 handle
168 .send(Op::SyncSession {
169 session_id: Some("edit-mode-test".to_string()),
170 messages: seeded_messages,
171 system_prompt: None,
172 system_prompt_override: false,
173 model: "deepseek-v4-pro".to_string(),
174 workspace: tmp.path().to_path_buf(),
175 mode: AppMode::Agent,
176 })
177 .await
178 .expect("sync session");
179 handle
180 .send(Op::ChangeMode {
181 mode: AppMode::Plan,
182 allow_shell: false,
183 trust_mode: false,
184 auto_approve: false,
185 approval_mode: ApprovalMode::Suggest,
186 configured_sandbox_mode: None,
187 })
188 .await
189 .expect("send plan mode");
190 handle
191 .send(Op::EditLastTurn {
192 new_message: "revise this in plan mode".to_string(),
193 submission_id: Some("sub-edit-1".to_string()),
194 })
195 .await
196 .expect("send edit");
197 // The replacement turn the edit replays must echo the edit's own
198 // correlation token, not the one of any earlier turn.
199 {
200 let mut rx = handle.rx_event.write().await;
201 loop {
202 let event = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
203 .await
204 .expect("timed out waiting for the edited turn")
205 .expect("engine event");
206 if let Event::TurnStarted { submission_id, .. } = event {
207 assert_eq!(
208 submission_id.as_deref(),
209 Some("sub-edit-1"),
210 "the edit's replacement turn must echo its correlation token"
211 );
212 break;
213 }
214 }
215 }
216
217 let (tx, rx) = tokio::sync::oneshot::channel();
218 handle
219 .send(Op::GetSessionSnapshot {
220 tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))),
221 })
222 .await
223 .expect("request snapshot");
224 let snapshot = tokio::time::timeout(model_turn_event_timeout(), rx)
225 .await
226 .expect("snapshot response")
227 .expect("snapshot");
228
229 assert_eq!(snapshot.mode, "plan");
230
231 let requests = server
232 .received_requests()
233 .await
234 .expect("recorded replacement request");
235 assert_eq!(
236 requests.len(),
237 1,
238 "edit must dispatch exactly one replacement turn"
239 );
240 handle.send(Op::Shutdown).await.expect("shutdown engine");
241 run.await.expect("engine task");
242 }
243
244 #[tokio::test]
245 #[allow(clippy::await_holding_lock)]
246 async fn edit_last_turn_cuts_at_user_prompt_before_tool_results() {
247 use wiremock::matchers::{method, path};
248 use wiremock::{Mock, MockServer, ResponseTemplate};
249
250 // Tool results persist with role "user"; the edit cut must land on the
251 // last genuine user prompt, not on the trailing tool_result of the
252 // previous turn.
253 let _lock = lock_test_env();
254 let tmp = tempdir().expect("tempdir");
255 let server = MockServer::start().await;
256 let done_sse = concat!(
257 "data: {\"id\":\"chatcmpl-edit-cut\",\"choices\":[{\"index\":0,",
258 "\"delta\":{\"content\":\"Revised answer.\"},\"finish_reason\":null}]}\n\n",
259 "data: {\"id\":\"chatcmpl-edit-cut\",\"choices\":[{\"index\":0,",
260 "\"delta\":{},\"finish_reason\":\"stop\"}]}\n\n",
261 "data: [DONE]\n\n",
262 );
263 Mock::given(method("POST"))
264 .and(path("/v1/chat/completions"))
265 .respond_with(
266 ResponseTemplate::new(200)
267 .insert_header("content-type", "text/event-stream")
268 .set_body_string(done_sse),
269 )
270 .expect(1)
271 .mount(&server)
272 .await;
273
274 let api_config = Config {
275 ..Config::default()
276 }
277 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
278 let config = EngineConfig {
279 workspace: tmp.path().to_path_buf(),
280 model: "deepseek-v4-pro".to_string(),
281 snapshots_enabled: false,
282 subagents_enabled: false,
283 ..Default::default()
284 };
285 let (engine, handle) = Engine::new(config, &api_config);
286
287 let run = tokio::spawn(engine.run());
288 let seeded_messages = vec![
289 Message {
290 role: Role::User,
291 content: vec![ContentBlock::Text {
292 text: "original prompt".to_string(),
293 cache_control: None,
294 }],
295 },
296 Message {
297 role: Role::Assistant,
298 content: vec![ContentBlock::ToolUse {
299 execution_id: None,
300 id: "call_1".to_string(),
301 name: "Bash".to_string(),
302 input: serde_json::json!({"command": "printf hi"}),
303 caller: None,
304 thought_signature: None,
305 }],
306 },
307 Message {
308 role: Role::User,
309 content: vec![ContentBlock::ToolResult {
310 execution_id: None,
311 tool_use_id: "call_1".to_string(),
312 content: "unique-tool-output-marker".to_string(),
313 is_error: None,
314 content_blocks: None,
315 }],
316 },
317 Message {
318 role: Role::Assistant,
319 content: vec![ContentBlock::Text {
320 text: "final answer".to_string(),
321 cache_control: None,
322 }],
323 },
324 ];
325 handle
326 .send(Op::SyncSession {
327 session_id: Some("edit-cut-test".to_string()),
328 messages: seeded_messages,
329 system_prompt: None,
330 system_prompt_override: false,
331 model: "deepseek-v4-pro".to_string(),
332 workspace: tmp.path().to_path_buf(),
333 mode: AppMode::Agent,
334 })
335 .await
336 .expect("sync session");
337 handle
338 .send(Op::EditLastTurn {
339 new_message: "edited prompt".to_string(),
340 submission_id: None,
341 })
342 .await
343 .expect("send edit");
344
345 // Ops are processed in order: once the snapshot arrives, the replacement
346 // turn has completed.
347 let (tx, rx) = tokio::sync::oneshot::channel();
348 handle
349 .send(Op::GetSessionSnapshot {
350 tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))),
351 })
352 .await
353 .expect("request snapshot");
354 let snapshot = tokio::time::timeout(model_turn_event_timeout(), rx)
355 .await
356 .expect("snapshot response")
357 .expect("snapshot");
358
359 assert_eq!(
360 snapshot.messages.len(),
361 2,
362 "the whole previous turn (prompt, tool_use, tool_result, answer) must be cut: {:?}",
363 snapshot.messages
364 );
365 assert!(
366 snapshot.messages.iter().all(|message| !message
367 .content
368 .iter()
369 .any(|block| matches!(block, ContentBlock::ToolResult { .. }))),
370 "no tool_result may survive the cut: {:?}",
371 snapshot.messages
372 );
373 let replacement_text = message_text_of(&snapshot.messages[0]);
374 assert!(
375 replacement_text.contains("edited prompt"),
376 "first surviving message is the edited prompt: {replacement_text}"
377 );
378
379 let requests = server
380 .received_requests()
381 .await
382 .expect("recorded replacement request");
383 assert_eq!(requests.len(), 1);
384 let body = String::from_utf8(requests[0].body.clone()).expect("request body utf8");
385 assert!(body.contains("edited prompt"));
386 assert!(
387 !body.contains("original prompt"),
388 "old prompt must not leak into the replacement turn: {body}"
389 );
390 assert!(
391 !body.contains("unique-tool-output-marker"),
392 "tool round-trip must not leak into the replacement turn: {body}"
393 );
394
395 handle.send(Op::Shutdown).await.expect("shutdown engine");
396 run.await.expect("engine task");
397 }
398
399 #[tokio::test]
400 #[allow(clippy::await_holding_lock)]
401 async fn edit_last_turn_without_user_prompt_errors_and_sends_nothing() {
402 use wiremock::matchers::{method, path};
403 use wiremock::{Mock, MockServer, ResponseTemplate};
404
405 let _lock = lock_test_env();
406 let tmp = tempdir().expect("tempdir");
407 let server = MockServer::start().await;
408 Mock::given(method("POST"))
409 .and(path("/v1/chat/completions"))
410 .respond_with(ResponseTemplate::new(200))
411 .expect(0)
412 .mount(&server)
413 .await;
414
415 let api_config = Config {
416 ..Config::default()
417 }
418 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
419 let config = EngineConfig {
420 workspace: tmp.path().to_path_buf(),
421 model: "deepseek-v4-pro".to_string(),
422 snapshots_enabled: false,
423 subagents_enabled: false,
424 ..Default::default()
425 };
426 let (engine, handle) = Engine::new(config, &api_config);
427
428 let run = tokio::spawn(engine.run());
429 // History without any genuine user prompt: nothing to edit. The engine
430 // must surface an error instead of silently appending the message.
431 handle
432 .send(Op::SyncSession {
433 session_id: Some("edit-no-user-test".to_string()),
434 messages: vec![Message {
435 role: Role::Assistant,
436 content: vec![ContentBlock::Text {
437 text: "assistant only".to_string(),
438 cache_control: None,
439 }],
440 }],
441 system_prompt: None,
442 system_prompt_override: false,
443 model: "deepseek-v4-pro".to_string(),
444 workspace: tmp.path().to_path_buf(),
445 mode: AppMode::Agent,
446 })
447 .await
448 .expect("sync session");
449 handle
450 .send(Op::EditLastTurn {
451 new_message: "edited prompt".to_string(),
452 submission_id: None,
453 })
454 .await
455 .expect("send edit");
456
457 let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
458 let mut saw_edit_error = false;
459 let mut saw_failed_terminal = false;
460 {
461 let mut events = handle.rx_event.write().await;
462 while let Ok(Some(event)) = tokio::time::timeout_at(deadline, events.recv()).await {
463 match event {
464 Event::Error { envelope, .. } => {
465 assert_eq!(envelope.code, "edit_last_turn_no_user_prompt");
466 assert!(!envelope.recoverable);
467 assert!(
468 envelope.message.contains("no user message"),
469 "unexpected error: {}",
470 envelope.message
471 );
472 saw_edit_error = true;
473 }
474 Event::TurnComplete { status, error, .. } => {
475 assert_eq!(status, TurnOutcomeStatus::Failed);
476 assert!(
477 error
478 .as_deref()
479 .is_some_and(|message| message.contains("no user message")),
480 "failed edit terminal must carry the rejection: {error:?}"
481 );
482 saw_failed_terminal = true;
483 break;
484 }
485 _ => {}
486 }
487 }
488 }
489 assert!(saw_edit_error, "edit without a user prompt must error out");
490 assert!(
491 saw_failed_terminal,
492 "edit rejection must complete the submitted host lifecycle"
493 );
494
495 let (tx, rx) = tokio::sync::oneshot::channel();
496 handle
497 .send(Op::GetSessionSnapshot {
498 tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))),
499 })
500 .await
501 .expect("request snapshot");
502 let snapshot = tokio::time::timeout(Duration::from_secs(2), rx)
503 .await
504 .expect("snapshot response")
505 .expect("snapshot");
506 assert_eq!(
507 snapshot.messages.len(),
508 1,
509 "failed edit must not append the new message: {:?}",
510 snapshot.messages
511 );
512
513 // An unsupported latest user turn is still a history boundary. It must
514 // fail in place rather than falling through to the older text prompt and
515 // deleting a larger portion of the conversation.
516 let image_only_history = vec![
517 Message {
518 role: Role::User,
519 content: vec![ContentBlock::Text {
520 text: "older editable prompt".to_string(),
521 cache_control: None,
522 }],
523 },
524 Message {
525 role: Role::Assistant,
526 content: vec![ContentBlock::Text {
527 text: "older response".to_string(),
528 cache_control: None,
529 }],
530 },
531 Message {
532 role: Role::User,
533 content: vec![ContentBlock::ImageUrl {
534 image_url: codewhale_models::ImageUrlContent {
535 url: "data:image/png;base64,AAAA".to_string(),
536 },
537 }],
538 },
539 ];
540 handle
541 .send(Op::SyncSession {
542 session_id: Some("edit-unsupported-user-test".to_string()),
543 messages: image_only_history.clone(),
544 system_prompt: None,
545 system_prompt_override: false,
546 model: "deepseek-v4-pro".to_string(),
547 workspace: tmp.path().to_path_buf(),
548 mode: AppMode::Agent,
549 })
550 .await
551 .expect("sync unsupported user session");
552 handle
553 .send(Op::EditLastTurn {
554 new_message: "must not replace the older prompt".to_string(),
555 submission_id: None,
556 })
557 .await
558 .expect("send unsupported edit");
559
560 let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
561 let mut saw_unsupported_error = false;
562 let mut saw_unsupported_terminal = false;
563 {
564 let mut events = handle.rx_event.write().await;
565 while let Ok(Some(event)) = tokio::time::timeout_at(deadline, events.recv()).await {
566 match event {
567 Event::Error { envelope, .. } => {
568 assert_eq!(envelope.code, "edit_last_turn_unsupported_user_content");
569 assert!(!envelope.recoverable);
570 saw_unsupported_error = true;
571 }
572 Event::TurnComplete { status, error, .. } => {
573 assert_eq!(status, TurnOutcomeStatus::Failed);
574 assert!(
575 error
576 .as_deref()
577 .is_some_and(|message| message
578 .contains("latest user message has no editable text")),
579 "unsupported edit terminal must carry the rejection: {error:?}"
580 );
581 saw_unsupported_terminal = true;
582 break;
583 }
584 _ => {}
585 }
586 }
587 }
588 assert!(saw_unsupported_error);
589 assert!(saw_unsupported_terminal);
590
591 let (tx, rx) = tokio::sync::oneshot::channel();
592 handle
593 .send(Op::GetSessionSnapshot {
594 tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))),
595 })
596 .await
597 .expect("request unsupported snapshot");
598 let unsupported_snapshot = tokio::time::timeout(Duration::from_secs(2), rx)
599 .await
600 .expect("unsupported snapshot response")
601 .expect("unsupported snapshot");
602 assert_eq!(
603 unsupported_snapshot.messages, image_only_history,
604 "unsupported latest user content must leave the entire history unchanged"
605 );
606
607 let requests = server.received_requests().await.expect("recorded requests");
608 assert!(
609 requests.is_empty(),
610 "failed edit must not dispatch a provider turn"
611 );
612
613 handle.send(Op::Shutdown).await.expect("shutdown engine");
614 run.await.expect("engine task");
615 }
616
617 #[tokio::test]
618 async fn provider_runtime_status_reports_configured_zai_cap_without_client() {
619 let (engine, handle) = {
620 let _lock = lock_test_env();
621 let _zai_key = EnvVarGuard::remove("ZAI_API_KEY");
622 let _zai_alt_key = EnvVarGuard::remove("Z_AI_API_KEY");
623 let api_config = Config {
624 provider: Some("zai".to_string()),
625 ..Config::default()
626 };
627 Engine::new(EngineConfig::default(), &api_config)
628 };
629
630 let run = tokio::spawn(engine.run());
631 let status = tokio::time::timeout(Duration::from_secs(2), handle.get_provider_runtime_status())
632 .await
633 .expect("provider runtime status response")
634 .expect("provider runtime status");
635
636 assert_eq!(status.provider, ProviderKind::Zai);
637 assert_eq!(
638 status.request_concurrency_limit,
639 Some(crate::config::DEFAULT_ZAI_PROVIDER_MAX_CONCURRENCY)
640 );
641 assert_eq!(status.active_provider_requests, 0);
642
643 run.abort();
644 }
645
646 #[test]
647 fn detects_context_length_errors_from_provider_payloads() {
648 let msg = r#"SSE stream request failed: HTTP 400 Bad Request: {"error":{"message":"This model's maximum context length is 131072 tokens. However, you requested 153056 tokens (148960 in the messages, 4096 in the completion).","type":"invalid_request_error"}}"#;
649 assert!(is_context_length_error_message(msg));
650 // llama.cpp's server wording (#6374): a genuine overflow on a local route
651 // must enter the bounded recovery path too.
652 assert!(is_context_length_error_message(
653 r#"SSE stream request failed: HTTP 400 Bad Request: {"error":{"code":400,"message":"the request exceeds the available context size. try increasing the context size or enable context shift","type":"invalid_request_error"}}"#
654 ));
655 assert!(!is_context_length_error_message(
656 "SSE stream request failed: HTTP 400 Bad Request: model not found"
657 ));
658 }
659
660 /// #6374: the exhausted-recovery message must name levers the reader has.
661 #[test]
662 fn context_overflow_exhausted_message_names_levers_that_exist_in_the_mode() {
663 let headless = super::context::context_overflow_exhausted_message(false, 2, 98_739, 97_280);
664 assert!(
665 !headless.contains("/compact") && !headless.contains("/clear"),
666 "a headless host has no command layer: {headless}"
667 );
668 assert!(
669 headless.contains("2 emergency compaction passes"),
670 "{headless}"
671 );
672 assert!(
673 headless.contains("CODEWHALE_MAX_OUTPUT_TOKENS"),
674 "{headless}"
675 );
676 let interactive = super::context::context_overflow_exhausted_message(true, 1, 98_739, 97_280);
677 assert!(
678 interactive.contains("/compact") && interactive.contains("/clear"),
679 "{interactive}"
680 );
681 assert!(
682 interactive.contains("1 emergency compaction pass "),
683 "{interactive}"
684 );
685 }
686
687 #[test]
688 fn context_budget_scenario() {
689 // Scenario consolidation of: context_budget_reserves_output_and_headroom, context_budget_uses_conservative_fallback_for_unknown_models, context_budget_uses_provider_effective_window_for_openai_codex
690 // from context_budget_reserves_output_and_headroom
691 {
692 // Serialize with other tests that mutate DEEPSEEK_MAX_OUTPUT_TOKENS so
693 // the internal effective_max_output_tokens() call sees a stable env.
694 let _lock = lock_test_env();
695 // Preflight reserves exactly the route-effective output request plus the
696 // shared safety headroom, even on a 1M route.
697 let budget = context_input_budget_for_provider(ProviderKind::Deepseek, "deepseek-v4-pro")
698 .expect("deepseek-v4-pro should have a known context window");
699 let v4_window: usize = 1_000_000;
700 let expected = v4_window
701 - effective_max_output_tokens_for_route(ProviderKind::Deepseek, "deepseek-v4-pro", None)
702 as usize
703 - 1_024usize;
704 assert_eq!(budget, expected);
705 }
706 // from context_budget_uses_conservative_fallback_for_unknown_models
707 {
708 let _lock = lock_test_env();
709 let budget = context_input_budget_for_provider(ProviderKind::Openai, "auto")
710 .expect("unknown/auto model ids should still get a conservative hard preflight budget");
711 let expected = 128_000usize
712 - effective_max_output_tokens_for_route(ProviderKind::Openai, "auto", None) as usize
713 - 1_024usize;
714 assert_eq!(budget, expected);
715 }
716 // from context_budget_uses_provider_effective_window_for_openai_codex
717 {
718 let _lock = lock_test_env();
719 let budget = context_input_budget_for_provider(ProviderKind::OpenaiCodex, "gpt-5.5")
720 .expect("OpenAI Codex should use a conservative fallback without route metadata");
721 let expected = usize::try_from(crate::config::OPENAI_CODEX_EFFECTIVE_CONTEXT_WINDOW_TOKENS)
722 .expect("context window fits usize")
723 - crate::config::provider_capability(ProviderKind::OpenaiCodex, "gpt-5.5")
724 .max_output
725 .expect("Codex route publishes a deliberate conservative output cap")
726 as usize
727 - 1_024usize;
728 assert_eq!(budget, expected);
729 }
730 }
731
732 #[test]
733 fn route_context_scenario() {
734 // Scenario consolidation of: route_context_budget_uses_shared_budget_service, route_context_budget_prefers_resolved_route_limits
735 // from route_context_budget_uses_shared_budget_service
736 {
737 let _lock = lock_test_env();
738 let budget =
739 route_context_budget_for_provider(ProviderKind::OpenaiCodex, "gpt-5.5", 380_000)
740 .expect("OpenAI Codex should produce a route budget");
741
742 assert_eq!(
743 budget.window_tokens,
744 u64::from(crate::config::OPENAI_CODEX_EFFECTIVE_CONTEXT_WINDOW_TOKENS)
745 );
746 assert_eq!(
747 budget.output_cap_tokens,
748 u64::from(
749 crate::config::provider_capability(ProviderKind::OpenaiCodex, "gpt-5.5")
750 .max_output
751 .expect("Codex route publishes a deliberate conservative output cap")
752 )
753 );
754 assert_eq!(
755 budget.pressure,
756 crate::context_budget::PressureLevel::Critical
757 );
758 assert!(!budget.fits_additional(1));
759 }
760 // from route_context_budget_prefers_resolved_route_limits
761 {
762 let _lock = lock_test_env();
763 let limits = codewhale_config::route::RouteLimits {
764 context_tokens: Some(128_000),
765 input_tokens: None,
766 output_tokens: Some(32_768),
767 };
768 let budget = route_context_budget_for_route(
769 ProviderKind::Openrouter,
770 "deepseek/deepseek-v4-pro",
771 Some(limits),
772 60_000,
773 )
774 .expect("route limits should produce a budget");
775
776 assert_eq!(budget.window_tokens, 128_000);
777 assert_eq!(budget.output_cap_tokens, 32_768);
778 assert_eq!(budget.available_input_tokens, 34_208);
779 }
780 }
781
782 #[test]
783 fn route_input_limit_blocks_oversized_preflight_before_transport() {
784 let _lock = lock_test_env();
785 let limits = codewhale_config::route::RouteLimits {
786 context_tokens: Some(1_000_000),
787 input_tokens: Some(128_000),
788 output_tokens: Some(64_000),
789 };
790 let estimated_input = 200_000;
791 let budget = route_context_budget_for_route(
792 ProviderKind::Vllm,
793 "DeepSeek-V4-Flash",
794 Some(limits),
795 estimated_input,
796 )
797 .expect("resolved route limits should produce the turn-loop preflight budget");
798
799 assert_eq!(budget.window_tokens, 1_000_000);
800 assert_eq!(budget.output_cap_tokens, 64_000);
801 assert_eq!(budget.input_budget_ceiling, 128_000);
802 assert_eq!(budget.available_input_tokens, 0);
803 assert!(
804 estimated_input > usize::try_from(budget.input_budget_ceiling).unwrap(),
805 "the turn-loop preflight must recover before constructing a network request"
806 );
807 }
808
809 /// #6374: the preflight guard measured a ×1.5-inflated estimate against the
810 /// honest input ceiling, so a route refused at two thirds of its budget with
811 /// the request never leaving the machine. The window here is calibrated so the
812 /// honest estimate sits below the ceiling and the inflated one above it; the
813 /// turn must reach the model with its history untouched.
814 #[tokio::test]
815 async fn preflight_guard_measures_honest_input_against_the_input_ceiling() {
816 let _lock = lock_test_env();
817 let _output_env = ScopedDeepSeekMaxOutputTokens::unset();
818 let workspace = tempdir().expect("workspace");
819 let _home = EnvVarGuard::set("CODEWHALE_HOME", workspace.path());
820 let mock = std::sync::Arc::new(crate::llm_client::mock::MockLlmClient::new(vec![
821 crate::llm_client::mock::canned::simple_text_turn("continuing"),
822 ]));
823 let (mut engine, _handle) = Engine::new_with_model_client(
824 EngineConfig {
825 terminal_chrome_enabled: false,
826 ..deterministic_engine_config(workspace.path())
827 },
828 &Config::default(),
829 mock.clone(),
830 );
831 // Only the preflight guard is under test; the auto-compaction gate stays out.
832 engine.config.compaction.enabled = false;
833 let history: Vec<Message> = [
834 (Role::User, "x".repeat(120_000)),
835 (Role::Assistant, "y".repeat(100_000)),
836 (Role::User, "please continue".to_string()),
837 ]
838 .into_iter()
839 .map(|(role, text)| Message {
840 role,
841 content: vec![ContentBlock::Text {
842 text,
843 cache_control: None,
844 }],
845 })
846 .collect();
847 for message in &history {
848 engine.session.add_message(message.clone());
849 }
850 let system = engine.session.system_prompt.clone();
851 let honest = crate::compaction::estimate_input_tokens_for_pressure(&history, system.as_ref());
852 let inflated = crate::compaction::estimate_input_tokens_conservative(&history, system.as_ref());
853 assert!(
854 inflated > honest + 20_000,
855 "fixture must separate the estimators: honest {honest}, inflated {inflated}"
856 );
857 let output_cap = 4_096u64;
858 let target_ceiling = u64::try_from((honest + inflated) / 2).unwrap();
859 engine.active_route_limits = Some(codewhale_config::route::RouteLimits {
860 context_tokens: Some(
861 target_ceiling + output_cap + crate::context_budget::CONTEXT_HEADROOM_TOKENS,
862 ),
863 input_tokens: None,
864 output_tokens: Some(output_cap),
865 });
866 let ceiling = route_context_budget_for_route(
867 engine.api_provider,
868 &engine.session.model,
869 engine.active_route_limits,
870 0,
871 )
872 .expect("route limits produce a budget")
873 .input_budget_ceiling;
874 let ceiling = usize::try_from(ceiling).unwrap();
875 assert!(
876 honest < ceiling && ceiling < inflated,
877 "calibration: honest {honest} < ceiling {ceiling} < inflated {inflated}"
878 );
879
880 let registry =
881 crate::tools::ToolRegistry::new(crate::tools::spec::ToolContext::new(workspace.path()));
882 let catalog = registry.to_api_tools_with_cache(true);
883 let surface = crate::core::engine::tool_catalog::ToolSurfacePolicy::new(
884 registry,
885 Some(catalog),
886 codewhale_config::AppMode::Agent,
887 &engine.config.tools_always_load,
888 &[],
889 false,
890 None,
891 None,
892 Some(4),
893 crate::core::engine::tool_catalog::ToolMode::Direct,
894 );
895 let (status, error) = engine
896 .run_turn(
897 &mut crate::core::turn::TurnContext::new(8),
898 surface,
899 None,
900 None,
901 )
902 .await;
903 assert_eq!(status, TurnOutcomeStatus::Completed, "{error:?}");
904 assert_eq!(
905 mock.call_count(),
906 1,
907 "the only model request is the turn itself, not an emergency compaction"
908 );
909 let request = mock.last_request().expect("the turn reached the model");
910 assert_eq!(
911 request.messages.len(),
912 history.len(),
913 "history reached the model without an emergency compaction pass"
914 );
915 }
916
917 #[test]
918 fn kimi_catalog_output_ceiling_does_not_collapse_input_budget() {
919 let _lock = lock_test_env();
920 let _guard = ScopedDeepSeekMaxOutputTokens::unset();
921 let documented =
922 route_context_budget_for_route(ProviderKind::Moonshot, "kimi-k2.7-code", None, 0)
923 .expect("bundled Kimi limits should produce a budget");
924 assert_eq!(documented.window_tokens, 262_144);
925 assert_eq!(documented.output_cap_tokens, 32_768);
926 assert_eq!(documented.available_input_tokens, 228_352);
927
928 // #4368/#4378: Models.dev may report Kimi's full 262K context as both its
929 // context window and provider output ceiling. That ceiling must not be
930 // reserved as though every normal turn requested 262K of output; the
931 // integrated Kimi route cap is 32K.
932 let limits = codewhale_config::route::RouteLimits {
933 context_tokens: Some(262_144),
934 input_tokens: None,
935 output_tokens: Some(262_144),
936 };
937
938 let budget =
939 route_context_budget_for_route(ProviderKind::Moonshot, "kimi-k2.7-code", Some(limits), 0)
940 .expect("Kimi route limits should produce a budget");
941
942 assert_eq!(budget.window_tokens, 262_144);
943 assert_eq!(budget.output_cap_tokens, 32_768);
944 assert_eq!(budget.available_input_tokens, 228_352);
945 }
946
947 #[test]
948 fn effective_max_scenario() {
949 // Scenario consolidation of: effective_max_output_tokens_for_route_caps_to_route_output_limit, effective_max_output_tokens_for_route_caps_to_context_window, effective_max_output_tokens_for_route_keeps_tiny_window_positive, effective_max_output_tokens_caps_api_request_for_large_window_models, effective_max_output_tokens_env_override_rejects_zero_and_invalid
950 // from effective_max_output_tokens_for_route_caps_to_route_output_limit
951 {
952 let _lock = lock_test_env();
953 let limits = codewhale_config::route::RouteLimits {
954 context_tokens: Some(1_000_000),
955 input_tokens: None,
956 output_tokens: Some(8_192),
957 };
958
959 assert_eq!(
960 effective_max_output_tokens_for_route(
961 ProviderKind::Deepseek,
962 "deepseek-v4-pro",
963 Some(limits),
964 ),
965 8_192
966 );
967 }
968 // from effective_max_output_tokens_for_route_caps_to_context_window
969 {
970 let _lock = lock_test_env();
971 let limits = codewhale_config::route::RouteLimits {
972 context_tokens: Some(32_000),
973 input_tokens: None,
974 output_tokens: None,
975 };
976
977 let cap = effective_max_output_tokens_for_route(
978 ProviderKind::Deepseek,
979 "deepseek-v4-pro",
980 Some(limits),
981 );
982
983 assert!(cap < 32_000, "request cap must fit the configured window");
984 assert!(
985 cap > 0,
986 "small configured windows should still allow output"
987 );
988 }
989 // from effective_max_output_tokens_for_route_keeps_tiny_window_positive
990 {
991 let _lock = lock_test_env();
992 let limits = codewhale_config::route::RouteLimits {
993 context_tokens: Some(2_048),
994 input_tokens: None,
995 output_tokens: None,
996 };
997
998 assert_eq!(
999 effective_max_output_tokens_for_route(
1000 ProviderKind::Deepseek,
1001 "deepseek-v4-pro",
1002 Some(limits),
1003 ),
1004 1
1005 );
1006 }
1007 // from effective_max_output_tokens_caps_api_request_for_large_window_models
1008 {
1009 // Serialize with other tests that mutate DEEPSEEK_MAX_OUTPUT_TOKENS so
1010 // v4_cap and flash_cap below see the same env state.
1011 let _lock = lock_test_env();
1012 // Hosted V4 documents a 384K capability ceiling in the bundled catalogue,
1013 // but a ceiling is not a safe no-config request size. The operator can
1014 // still request a larger value explicitly; the automatic request starts
1015 // at the ordinary 64K cap (#5516/#5518).
1016 let v4_cap = effective_max_output_tokens("deepseek-v4-pro");
1017 assert_eq!(
1018 v4_cap, 65_536,
1019 "hosted V4 must not turn the 384K capability maximum into the default request, got {v4_cap}"
1020 );
1021
1022 let flash_cap = effective_max_output_tokens("deepseek-v4-flash");
1023 assert_eq!(v4_cap, flash_cap);
1024 }
1025 // from effective_max_output_tokens_env_override_rejects_zero_and_invalid
1026 {
1027 let _lock = lock_test_env();
1028 // Establish the heuristic baseline with the env unset.
1029 let baseline = {
1030 let _guard = ScopedDeepSeekMaxOutputTokens::unset();
1031 effective_max_output_tokens("deepseek-v4-pro")
1032 };
1033 assert!(baseline > 0);
1034
1035 // 0, non-numeric, and empty values must all fall through to the heuristic
1036 // rather than producing a zero/garbage cap that would silently break
1037 // request budgeting.
1038 for raw in ["0", "abc", "", " ", "-1"] {
1039 let _guard = ScopedDeepSeekMaxOutputTokens::set(raw);
1040 assert_eq!(
1041 effective_max_output_tokens("deepseek-v4-pro"),
1042 baseline,
1043 "env={raw:?} should fall through to heuristic"
1044 );
1045 }
1046 }
1047 }
1048
1049 #[test]
1050 fn codex_route_without_output_metadata_uses_oauth_capability_floor() {
1051 let _lock = lock_test_env();
1052 let limits = codewhale_config::route::RouteLimits {
1053 context_tokens: Some(272_000),
1054 input_tokens: None,
1055 output_tokens: None,
1056 };
1057
1058 assert_eq!(
1059 effective_max_output_tokens_for_route(ProviderKind::OpenaiCodex, "gpt-5.5", Some(limits)),
1060 4_096
1061 );
1062 let budget =
1063 route_context_budget_for_route(ProviderKind::OpenaiCodex, "gpt-5.5", Some(limits), 0)
1064 .expect("Codex route budget");
1065 assert_eq!(budget.output_cap_tokens, 4_096);
1066 }
1067
1068 #[test]
1069 fn reasoning_max_does_not_add_a_second_deepseek_v4_output_reservation() {
1070 let _lock = lock_test_env();
1071 let _codewhale = EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS");
1072 let _deepseek = EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS");
1073 let limits = codewhale_config::route::RouteLimits {
1074 context_tokens: Some(327_680),
1075 input_tokens: None,
1076 output_tokens: None,
1077 };
1078 let cap = effective_max_output_tokens_for_route(
1079 ProviderKind::Vllm,
1080 "DeepSeek-V4-Flash",
1081 Some(limits),
1082 );
1083 let request = codewhale_core::request::prepare_primary_turn_request(
1084 codewhale_core::request::PrimaryTurnRequest {
1085 model: "DeepSeek-V4-Flash".to_string(),
1086 messages: Vec::new(),
1087 max_tokens: cap,
1088 system: None,
1089 tools: None,
1090 tool_choice: None,
1091 reasoning_effort: Some("max".to_string()),
1092 },
1093 );
1094 let budget = route_context_budget_for_route(
1095 ProviderKind::Vllm,
1096 "DeepSeek-V4-Flash",
1097 Some(limits),
1098 105_000,
1099 )
1100 .expect("max-reasoning vLLM route budget");
1101
1102 assert_eq!(request.reasoning_effort.as_deref(), Some("max"));
1103 assert_eq!(request.max_tokens, 65_536);
1104 assert_eq!(budget.output_cap_tokens, u64::from(request.max_tokens));
1105 assert_eq!(budget.input_budget_ceiling, 261_120);
1106 assert!(budget.available_input_tokens > 0);
1107 }
1108
1109 struct ScopedDeepSeekMaxOutputTokens {
1110 previous: Option<OsString>,
1111 }
1112
1113 impl ScopedDeepSeekMaxOutputTokens {
1114 fn set(value: &str) -> Self {
1115 let previous = std::env::var_os("DEEPSEEK_MAX_OUTPUT_TOKENS");
1116 // Safety: tests using this helper serialize with lock_test_env() and
1117 // restore the original value in Drop.
1118 unsafe {
1119 std::env::set_var("DEEPSEEK_MAX_OUTPUT_TOKENS", value);
1120 }
1121 Self { previous }
1122 }
1123
1124 fn unset() -> Self {
1125 let previous = std::env::var_os("DEEPSEEK_MAX_OUTPUT_TOKENS");
1126 // Safety: see set().
1127 unsafe {
1128 std::env::remove_var("DEEPSEEK_MAX_OUTPUT_TOKENS");
1129 }
1130 Self { previous }
1131 }
1132 }
1133
1134 impl Drop for ScopedDeepSeekMaxOutputTokens {
1135 fn drop(&mut self) {
1136 // Safety: tests using this helper serialize with lock_test_env().
1137 unsafe {
1138 if let Some(previous) = self.previous.take() {
1139 std::env::set_var("DEEPSEEK_MAX_OUTPUT_TOKENS", previous);
1140 } else {
1141 std::env::remove_var("DEEPSEEK_MAX_OUTPUT_TOKENS");
1142 }
1143 }
1144 }
1145 }
1146
1147 #[test]
1148 fn effective_max_output_tokens_env_override_returns_positive_value() {
1149 let _lock = lock_test_env();
1150 let _guard = ScopedDeepSeekMaxOutputTokens::set("16384");
1151
1152 // Override applies regardless of model — V4 hosted, V4 flash, and
1153 // self-hosted routes all return the env value verbatim before route clamps.
1154 assert_eq!(effective_max_output_tokens("deepseek-v4-pro"), 16_384);
1155 assert_eq!(effective_max_output_tokens("deepseek-v4-flash"), 16_384);
1156 assert_eq!(effective_max_output_tokens("qwen3-32b-256k"), 16_384);
1157 }
1158
1159 #[test]
1160 fn internal_context_budget_uses_the_wire_cap_across_window_sizes() {
1161 // Serialize with other tests that mutate DEEPSEEK_MAX_OUTPUT_TOKENS so
1162 // both branches below see a stable env.
1163 let _lock = lock_test_env();
1164 // Large routes use the same effective output cap that reaches the wire.
1165 let internal_budget =
1166 context_input_budget_for_provider(ProviderKind::Deepseek, "deepseek-v4-pro")
1167 .expect("V4 should have a known context window");
1168 let v4_window: usize = 1_000_000;
1169 let expected_internal = v4_window
1170 - effective_max_output_tokens_for_route(ProviderKind::Deepseek, "deepseek-v4-pro", None)
1171 as usize
1172 - 1_024usize;
1173 assert_eq!(internal_budget, expected_internal);
1174
1175 // A 256K self-hosted deployment uses the same rule and yields a usable
1176 // positive budget rather than silently disabling preflight/recovery.
1177 let small_window_budget =
1178 context_input_budget_for_provider(ProviderKind::Openai, "qwen3-32b-256k")
1179 .expect("a 256K-suffix model must yield Some budget via the effective-cap branch");
1180 let effective_output =
1181 effective_max_output_tokens_for_route(ProviderKind::Openai, "qwen3-32b-256k", None)
1182 as usize;
1183 let expected_small = 256_000 - effective_output - 1_024;
1184 assert_eq!(small_window_budget, expected_small);
1185 }
1186
1187 const ROUTE_128K: &str = "deepseek-v3.2-128k";
1188 const SESSION_6508: &str = "session-6508";
1189
1190 fn budget_128k() -> usize {
1191 crate::route_budget::route_inline_char_budget_for_route(
1192 ProviderKind::Deepseek,
1193 ROUTE_128K,
1194 None,
1195 )
1196 }
1197
1198 fn view_128k(tool_name: &str, output: &ToolResult) -> super::context::ToolResultContextView {
1199 super::context::tool_result_context_view(
1200 ProviderKind::Deepseek,
1201 ROUTE_128K,
1202 None,
1203 tool_name,
1204 output,
1205 )
1206 }
1207
1208 /// Run `f` with the spillover and session-artifact roots under a temp home.
1209 fn with_artifact_home<R>(f: impl FnOnce(&Path) -> R) -> R {
1210 let _spill_guard = crate::tools::truncate::TEST_SPILLOVER_GUARD
1211 .lock()
1212 .unwrap_or_else(|err| err.into_inner());
1213 let home = tempdir().expect("tempdir");
1214 let path = home.path().to_path_buf();
1215 crate::tools::truncate::with_test_home(&path, || f(&path))
1216 }
1217
1218 fn session_artifact_files(home: &Path) -> Vec<PathBuf> {
1219 let dir = home
1220 .join(".codewhale")
1221 .join("sessions")
1222 .join(SESSION_6508)
1223 .join("artifacts");
1224 fs::read_dir(dir)
1225 .map(|entries| entries.flatten().map(|entry| entry.path()).collect())
1226 .unwrap_or_default()
1227 }
1228
1229 #[test]
1230 fn under_budget_results_pass_through_whole_for_every_tool() {
1231 // #6508: one budget, sized by the route, decides what the model sees.
1232 // There is no per-tool-name soft limit: a web answer or a shell log that
1233 // fits the budget reaches the model byte for byte.
1234 // 3% of a 128K window is 15,360 characters (an operator opt-in can only
1235 // raise it; see route_budget's tests for the exact values).
1236 assert!(budget_128k() >= 15_360);
1237 let content = "w".repeat(14_000);
1238 let output = ToolResult::success(content.clone());
1239 for tool_name in [
1240 "exec_shell",
1241 "web_search",
1242 "Web",
1243 "web.run",
1244 "fetch_url",
1245 "read_file",
1246 "run_tests",
1247 ] {
1248 let view = view_128k(tool_name, &output);
1249 assert_eq!(view.text, content, "{tool_name} was cut under the budget");
1250 assert!(!view.needs_full_output_artifact);
1251 }
1252 }
1253
1254 #[test]
1255 fn over_budget_result_without_a_saved_copy_says_so_and_asks_for_one() {
1256 // The view never writes. Without a saved copy it asks the engine for
1257 // one, and until then it promises no ref it cannot honour.
1258 let raw = "shell line\n".repeat(4_000);
1259 let output = ToolResult::success(raw.clone());
1260 let view = view_128k("exec_shell", &output);
1261
1262 assert!(view.needs_full_output_artifact);
1263 assert!(view.text.chars().count() <= budget_128k());
1264 assert!(view.text.contains("the full output could not be saved"));
1265 assert!(view.text.contains("no tool call reaches this copy"));
1266 assert!(!view.text.contains("retrieve_tool_result"));
1267 }
1268
1269 #[test]
1270 fn oversized_tool_output_is_recoverable_before_serial_and_parallel_fanout() {
1271 use crate::llm_client::mock::{MockLlmClient, canned};
1272 use crate::tools::spec::{ToolCapability, ToolSpec};
1273
1274 struct OutputTool {
1275 parallel: bool,
1276 content: String,
1277 }
1278 #[async_trait::async_trait]
1279 impl ToolSpec for OutputTool {
1280 fn name(&self) -> &str {
1281 "fixture_output"
1282 }
1283 fn description(&self) -> &str {
1284 "Return output below the spill threshold."
1285 }
1286 fn input_schema(&self) -> Value {
1287 json!({"type": "object"})
1288 }
1289 fn capabilities(&self) -> Vec<ToolCapability> {
1290 vec![ToolCapability::ReadOnly]
1291 }
1292 fn supports_parallel(&self) -> bool {
1293 self.parallel
1294 }
1295 async fn execute(&self, _: Value, _: &ToolContext) -> Result<ToolResult, ToolError> {
1296 Ok(ToolResult::success(self.content.clone()))
1297 }
1298 }
1299
1300 with_artifact_home(|home| {
1301 tokio::runtime::Builder::new_current_thread()
1302 .enable_all()
1303 .build()
1304 .unwrap()
1305 .block_on(async {
1306 let raw = format!("{}MIDDLE{}", "h".repeat(5_000), "t".repeat(5_000));
1307 assert!(raw.len() < crate::tools::truncate::SPILLOVER_THRESHOLD_BYTES);
1308 for (parallel, count) in [(true, 1), (true, 2), (false, 1)] {
1309 let calls = [
1310 ("call-one", "fixture_output", "{}"),
1311 ("call-two", "fixture_output", "{}"),
1312 ];
1313 let mock = Arc::new(MockLlmClient::new(vec![
1314 tool_batch_turn(&calls[..count]),
1315 canned::simple_text_turn("done"),
1316 ]));
1317 let (mut engine, handle) = Engine::new_with_model_client(
1318 deterministic_engine_config(home),
1319 &Config::default(),
1320 mock.clone(),
1321 );
1322 engine.active_route_limits = Some(codewhale_config::route::RouteLimits {
1323 context_tokens: Some(64_000),
1324 input_tokens: None,
1325 output_tokens: Some(4_096),
1326 });
1327 let mut registry = crate::tools::ToolRegistry::new(ToolContext::new(home));
1328 registry.register(Arc::new(OutputTool {
1329 parallel,
1330 content: raw.clone(),
1331 }));
1332 let tools = Some(registry.to_api_tools_with_cache(true));
1333 let surface = test_tool_surface(&engine, registry, tools, AppMode::Agent);
1334 let mut turn = crate::core::turn::TurnContext::new(4);
1335 let (status, error) = engine.run_turn(&mut turn, surface, None, None).await;
1336 assert_eq!(status, TurnOutcomeStatus::Completed, "{error:?}");
1337 let requests = mock.captured_requests();
1338 assert_eq!(requests.len(), 2);
1339 let mut events = handle.rx_event.write().await;
1340 let mut completed = 0;
1341 while let Ok(event) = events.try_recv() {
1342 let Event::ToolCallComplete {
1343 result, model_call, ..
1344 } = event
1345 else {
1346 continue;
1347 };
1348 let output = result.expect("tool succeeded");
1349 assert_eq!(output.content, raw, "UI keeps the complete output");
1350 let metadata = output
1351 .metadata
1352 .expect("output must be preserved before fanout");
1353 let path = metadata["artifact_path"].as_str().expect("artifact path");
1354 assert_eq!(fs::read_to_string(path).unwrap(), raw);
1355 let reference = metadata["artifact_id"].as_str().unwrap();
1356 let results =
1357 guardian_tool_results(&requests[1], &model_call.unwrap().provider_id);
1358 assert_eq!(results.len(), 1);
1359 let (text, _) = results[0];
1360 assert!(text.len() <= 7_680, "64K route inline budget");
1361 assert!(!text.contains("MIDDLE"));
1362 assert!(text.contains("retrieve_tool_result"));
1363 assert!(text.contains(reference), "model and UI share the artifact");
1364 completed += 1;
1365 }
1366 assert_eq!(completed, count);
1367 }
1368 });
1369 });
1370 }
1371
1372 #[test]
1373 fn over_budget_result_writes_the_full_output_and_names_its_ref() {
1374 with_artifact_home(|home| {
1375 let raw = format!("FIRST LINE\n{}LAST LINE", "shell line\n".repeat(4_000));
1376 let mut output = ToolResult::success(raw.clone());
1377 assert!(view_128k("exec_shell", &output).needs_full_output_artifact);
1378
1379 assert!(
1380 crate::tools::truncate::preserve_full_output_for_model_context(
1381 &mut output,
1382 "call-over",
1383 "exec_shell",
1384 SESSION_6508,
1385 )
1386 );
1387 // The UI cell keeps the whole result; only metadata changed.
1388 assert_eq!(output.content, raw);
1389
1390 let view = view_128k("exec_shell", &output);
1391 assert!(!view.needs_full_output_artifact);
1392 assert!(view.text.chars().count() <= budget_128k());
1393 assert!(view.text.starts_with("FIRST LINE"));
1394 assert!(view.text.ends_with("LAST LINE"));
1395 assert!(view.text.contains("omitted range recovery:"));
1396 assert!(view.text.contains("ref=\"art_call-over\""));
1397
1398 let files = session_artifact_files(home);
1399 assert_eq!(files.len(), 1);
1400 assert_eq!(fs::read_to_string(&files[0]).expect("artifact"), raw);
1401 });
1402 }
1403
1404 #[test]
1405 fn spilled_preview_is_refit_not_recut() {
1406 // Spillover already saved the full output and left a ~40 KB preview. The
1407 // model view re-fits that preview to the budget with the same ref, and no
1408 // second artifact (which could only hold the preview) is written.
1409 with_artifact_home(|home| {
1410 let raw = format!(
1411 "HEAD START\n{}TAIL END",
1412 "spilled output line\n".repeat(16_000)
1413 );
1414 let mut output = ToolResult::success(raw.clone());
1415 assert!(
1416 crate::tools::truncate::apply_spillover_with_artifact(
1417 &mut output,
1418 "call-spill",
1419 "exec_shell",
1420 SESSION_6508,
1421 )
1422 .is_some()
1423 );
1424 assert_eq!(session_artifact_files(home).len(), 1);
1425
1426 let view = view_128k("exec_shell", &output);
1427 assert!(!view.needs_full_output_artifact);
1428 assert!(view.text.chars().count() <= budget_128k());
1429 assert!(view.text.starts_with("HEAD START"));
1430 assert!(view.text.ends_with("TAIL END"));
1431 assert_eq!(
1432 view.text.matches("omitted range recovery:").count(),
1433 1,
1434 "exactly one footer: {}",
1435 &view.text[..200]
1436 );
1437 assert!(view.text.contains("ref=\"art_call-spill\""));
1438 assert_eq!(session_artifact_files(home).len(), 1);
1439 });
1440 }
1441
1442 #[test]
1443 fn legacy_spill_without_ref_says_no_tool_call_reaches_it() {
1444 let head = "h".repeat(32 * 1024);
1445 let tail = "t".repeat(8 * 1024);
1446 let preview = format!("{head}\n\n… footer …\n\n…\n{tail}");
1447 let output = ToolResult::success(preview).with_metadata(json!({
1448 "spillover_path": "/tmp/tool_outputs/call-legacy.txt",
1449 "retained_head_bytes": head.len(),
1450 "retained_tail_bytes": tail.len(),
1451 "original_byte_count": 300_000,
1452 "truncated": true
1453 }));
1454
1455 let view = view_128k("exec_shell", &output);
1456 assert!(!view.needs_full_output_artifact);
1457 assert!(view.text.chars().count() <= budget_128k());
1458 assert!(view.text.contains("/tmp/tool_outputs/call-legacy.txt"));
1459 assert!(view.text.contains("no tool call reaches this copy"));
1460 assert!(!view.text.contains("retrieve_tool_result"));
1461 }
1462
1463 #[test]
1464 fn structured_run_tests_summary_is_recoverable_and_leads_with_failures() {
1465 with_artifact_home(|home| {
1466 let stdout = format!(
1467 "{}test result: FAILED. 1 failed",
1468 "test ok ... ok\n".repeat(3_000)
1469 );
1470 let raw = json!({
1471 "success": false,
1472 "exit_code": 101,
1473 "stdout": stdout,
1474 "stderr": "",
1475 "command": "(cd /repo && cargo test)"
1476 })
1477 .to_string();
1478 let mut output = ToolResult::success(raw.clone()).with_metadata(json!({
1479 "summary": "1 test failed: tools::git::tests::diff_keeps_the_last_file"
1480 }));
1481 assert!(view_128k("run_tests", &output).needs_full_output_artifact);
1482 assert!(
1483 crate::tools::truncate::preserve_full_output_for_model_context(
1484 &mut output,
1485 "call-tests",
1486 "run_tests",
1487 SESSION_6508,
1488 )
1489 );
1490
1491 let view = view_128k("run_tests", &output);
1492 assert!(!view.needs_full_output_artifact);
1493 assert!(view.text.chars().count() <= budget_128k());
1494 let failures = view
1495 .text
1496 .find("failure summary: 1 test failed: tools::git::tests::diff_keeps_the_last_file")
1497 .expect("failure summary inline");
1498 assert!(failures < view.text.find("stdout:").expect("stdout"));
1499 assert!(view.text.contains("test result: FAILED"));
1500 assert!(view.text.contains("ref=\"art_call-tests\""));
1501
1502 let files = session_artifact_files(home);
1503 assert_eq!(files.len(), 1);
1504 assert_eq!(fs::read_to_string(&files[0]).expect("artifact"), raw);
1505 });
1506 }
1507
1508 #[test]
1509 fn display_compaction_never_writes_artifacts() {
1510 // The TUI builds its API-message copy with the same pure view.
1511 with_artifact_home(|home| {
1512 let output = ToolResult::success("x".repeat(60_000));
1513 let text = compact_tool_result_for_route(
1514 ProviderKind::Deepseek,
1515 ROUTE_128K,
1516 None,
1517 "exec_shell",
1518 &output,
1519 );
1520 assert!(text.chars().count() <= budget_128k());
1521 assert!(session_artifact_files(home).is_empty());
1522 });
1523 }
1524
1525 #[test]
1526 fn evidence_bounded_preview_is_not_recompacted() {
1527 // The adaptive evidence envelope already produced an honest bounded
1528 // preview (head + footer with the recovery path + tail). The context
1529 // compactor must pass it through untouched, even beyond the 12K hard
1530 // limit — re-compacting would strip the recovery contract.
1531 let content = format!(
1532 "{}\n\n… 19.0 KiB of output omitted (123 lines) — full output at /tmp/art_call.txt; read it back with the read_file tool or with sed line ranges\n\n…\n{}",
1533 "h".repeat(32 * 1024),
1534 "t".repeat(8 * 1024)
1535 );
1536 let output = ToolResult::success(content.clone()).with_metadata(json!({
1537 "evidence_available": true,
1538 "truncated": true,
1539 "spillover_path": "/tmp/art_call.txt"
1540 }));
1541
1542 let context = compact_tool_result_for_context("deepseek-v3.2-128k", "Bash", &output);
1543 assert_eq!(context, content);
1544 assert!(context.contains("full output at /tmp/art_call.txt"));
1545 }
1546
1547 #[test]
1548 fn budgeted_read_result_is_not_truncated_a_second_time_by_the_context_compactor() {
1549 // C05: `read` bounds itself to an explicit per-call byte budget. The 12K
1550 // context hard limit used to re-truncate that bounded result into a 900-
1551 // char snippet, discarding both the content and the continuation footer.
1552 let content = format!(
1553 "{}\n\n[Showing lines 1-100 of 2000 (100000-byte output budget). Use offset=101 to continue.]",
1554 "r".repeat(90_000)
1555 );
1556 let budgeted = ToolResult::success(content.clone()).with_metadata(json!({
1557 "evidence_routing": "inline",
1558 "read_budget_bytes": 100_000
1559 }));
1560 let passed_through = compact_tool_result_for_context("deepseek-v3.2-128k", "read", &budgeted);
1561 assert_eq!(passed_through, content);
1562 assert!(passed_through.contains("Use offset=101 to continue"));
1563
1564 // The same bytes without a declared budget still take the ordinary path,
1565 // which is what proves the metadata (not the tool name) did the work.
1566 let unbudgeted = ToolResult::success(content.clone());
1567 let compacted = compact_tool_result_for_context("deepseek-v3.2-128k", "read", &unbudgeted);
1568 assert!(compacted.contains(crate::tools::truncate::SPILLOVER_RECOVERY_HINT));
1569 assert!(compacted.len() < content.len());
1570
1571 // A result that overran its own declared budget is not exempt.
1572 let overrun = ToolResult::success(content).with_metadata(json!({
1573 "read_budget_bytes": 1_000
1574 }));
1575 let compacted_overrun = compact_tool_result_for_context("deepseek-v3.2-128k", "read", &overrun);
1576 assert!(compacted_overrun.contains(crate::tools::truncate::SPILLOVER_RECOVERY_HINT));
1577 }
1578
1579 #[test]
1580 fn codex_tool_retention_uses_oauth_route_window_not_asmall_contract_model_window() {
1581 let content = "route-effective context\n".repeat(900);
1582 let output = ToolResult::success(content.clone());
1583 let limits = codewhale_config::route::RouteLimits {
1584 context_tokens: Some(272_000),
1585 input_tokens: None,
1586 output_tokens: None,
1587 };
1588
1589 // The budget follows the route's 272K window (3% of it, 32,640
1590 // characters), so this 21.6K result reaches the model whole.
1591 assert!(
1592 crate::route_budget::route_inline_char_budget_for_route(
1593 ProviderKind::OpenaiCodex,
1594 "gpt-5.5",
1595 Some(limits),
1596 ) >= 32_640
1597 );
1598 let context = compact_tool_result_for_route(
1599 ProviderKind::OpenaiCodex,
1600 "gpt-5.5",
1601 Some(limits),
1602 "read_file",
1603 &output,
1604 );
1605
1606 assert_eq!(context, content.trim());
1607 }
1608
1608 lines RUST