| 1 | #[tokio::test] |
| 2 | async fn sync_session_projects_persisted_subagent_handoff_for_headless_restore() { |
| 3 | let tmp = tempdir().expect("tempdir"); |
| 4 | let config = EngineConfig { |
| 5 | workspace: tmp.path().to_path_buf(), |
| 6 | model: "deepseek-v4-pro".to_string(), |
| 7 | ..Default::default() |
| 8 | }; |
| 9 | let (engine, handle) = Engine::new(config, &Config::default()); |
| 10 | let payload = concat!( |
| 11 | "Child result retained.\nCheckpoint: engine restore is covered.\n", |
| 12 | "<codewhale:subagent.done>{\"agent_id\":\"agent_headless\",", |
| 13 | "\"status\":\"completed\",\"summary_location\":\"previous_line\"}", |
| 14 | "</codewhale:subagent.done>", |
| 15 | ); |
| 16 | let messages = vec![ |
| 17 | Message { |
| 18 | role: Role::User, |
| 19 | content: vec![ContentBlock::Text { |
| 20 | text: "Keep the original task".to_string(), |
| 21 | cache_control: None, |
| 22 | }], |
| 23 | }, |
| 24 | crate::runtime_handoff::subagent_completion_runtime_message(payload), |
| 25 | ]; |
| 26 | |
| 27 | let run = tokio::spawn(engine.run()); |
| 28 | handle |
| 29 | .send(Op::SyncSession { |
| 30 | session_id: Some("headless-resume".to_string()), |
| 31 | messages, |
| 32 | system_prompt: None, |
| 33 | system_prompt_override: false, |
| 34 | model: "deepseek-v4-pro".to_string(), |
| 35 | workspace: tmp.path().to_path_buf(), |
| 36 | mode: AppMode::Agent, |
| 37 | }) |
| 38 | .await |
| 39 | .expect("sync session"); |
| 40 | |
| 41 | let (tx, rx) = tokio::sync::oneshot::channel(); |
| 42 | handle |
| 43 | .send(Op::GetSessionSnapshot { |
| 44 | tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))), |
| 45 | }) |
| 46 | .await |
| 47 | .expect("request snapshot"); |
| 48 | let snapshot = tokio::time::timeout(Duration::from_secs(2), rx) |
| 49 | .await |
| 50 | .expect("snapshot response") |
| 51 | .expect("snapshot"); |
| 52 | |
| 53 | assert_eq!(snapshot.messages.len(), 2); |
| 54 | assert!(snapshot.messages[0].content.iter().any( |
| 55 | |block| matches!(block, ContentBlock::Text { text, .. } if text == "Keep the original task") |
| 56 | )); |
| 57 | let restored = |
| 58 | crate::runtime_handoff::restored_subagent_checkpoint_display(&snapshot.messages[1]) |
| 59 | .expect("projected headless checkpoint"); |
| 60 | assert!(restored.contains("agent_headless")); |
| 61 | assert!(restored.contains("Checkpoint: engine restore is covered.")); |
| 62 | assert!(!restored.contains("runtime_event")); |
| 63 | assert!(!restored.contains("subagent.done")); |
| 64 | |
| 65 | run.abort(); |
| 66 | } |
| 67 | |
| 68 | #[tokio::test] |
| 69 | async fn session_snapshot_records_the_literal_custom_table_id() { |
| 70 | let tmp = tempdir().expect("tempdir"); |
| 71 | let api_config = crate::config::parse_config_base( |
| 72 | r#"provider = "custom" |
| 73 | [providers.custom] |
| 74 | kind = "openai-compatible" |
| 75 | base_url = "http://127.0.0.1:18180/v1" |
| 76 | model = "legacy-root-model" |
| 77 | auth_mode = "none" |
| 78 | "#, |
| 79 | ) |
| 80 | .expect("canonical literal custom table"); |
| 81 | let config = EngineConfig { |
| 82 | workspace: tmp.path().to_path_buf(), |
| 83 | model: "legacy-root-model".to_string(), |
| 84 | ..Default::default() |
| 85 | }; |
| 86 | let (engine, handle) = Engine::new(config, &api_config); |
| 87 | |
| 88 | let run = tokio::spawn(engine.run()); |
| 89 | let (tx, rx) = tokio::sync::oneshot::channel(); |
| 90 | handle |
| 91 | .send(Op::GetSessionSnapshot { |
| 92 | tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))), |
| 93 | }) |
| 94 | .await |
| 95 | .expect("request snapshot"); |
| 96 | let snapshot = tokio::time::timeout(Duration::from_secs(2), rx) |
| 97 | .await |
| 98 | .expect("snapshot response") |
| 99 | .expect("snapshot"); |
| 100 | |
| 101 | // The literal route is the `[providers.custom]` table since #6394. |
| 102 | assert_eq!(snapshot.model_provider, "custom"); |
| 103 | assert_eq!(snapshot.model_provider_id.as_deref(), Some("custom")); |
| 104 | run.abort(); |
| 105 | } |
| 106 | |
| 107 | #[tokio::test] |
| 108 | #[allow(clippy::await_holding_lock)] |
| 109 | async fn edit_last_turn_preserves_current_mode() { |
| 110 | use wiremock::matchers::{method, path}; |
| 111 | use wiremock::{Mock, MockServer, ResponseTemplate}; |
| 112 | |
| 113 | // EditLastTurn dispatches a real replacement turn. Pin that turn to a |
| 114 | // local, completing SSE response instead of depending on whichever |
| 115 | // provider configuration or network state the parallel test process has. |
| 116 | let _lock = lock_test_env(); |
| 117 | let tmp = tempdir().expect("tempdir"); |
| 118 | let server = MockServer::start().await; |
| 119 | let done_sse = concat!( |
| 120 | "data: {\"id\":\"chatcmpl-edit-mode\",\"choices\":[{\"index\":0,", |
| 121 | "\"delta\":{\"content\":\"Revised plan.\"},\"finish_reason\":null}]}\n\n", |
| 122 | "data: {\"id\":\"chatcmpl-edit-mode\",\"choices\":[{\"index\":0,", |
| 123 | "\"delta\":{},\"finish_reason\":\"stop\"}]}\n\n", |
| 124 | "data: [DONE]\n\n", |
| 125 | ); |
| 126 | Mock::given(method("POST")) |
| 127 | .and(path("/v1/chat/completions")) |
| 128 | .respond_with( |
| 129 | ResponseTemplate::new(200) |
| 130 | .insert_header("content-type", "text/event-stream") |
| 131 | .set_body_string(done_sse), |
| 132 | ) |
| 133 | .expect(1) |
| 134 | .mount(&server) |
| 135 | .await; |
| 136 | |
| 137 | let api_config = Config { |
| 138 | ..Config::default() |
| 139 | } |
| 140 | .with_legacy_root(Some("test-key".to_string()), Some(server.uri())); |
| 141 | let config = EngineConfig { |
| 142 | workspace: tmp.path().to_path_buf(), |
| 143 | model: "deepseek-v4-pro".to_string(), |
| 144 | snapshots_enabled: false, |
| 145 | subagents_enabled: false, |
| 146 | ..Default::default() |
| 147 | }; |
| 148 | let (engine, handle) = Engine::new(config, &api_config); |
| 149 | |
| 150 | let run = tokio::spawn(engine.run()); |
| 151 | let seeded_messages = vec![ |
| 152 | Message { |
| 153 | role: Role::User, |
| 154 | content: vec![ContentBlock::Text { |
| 155 | text: "draft the plan".to_string(), |
| 156 | cache_control: None, |
| 157 | }], |
| 158 | }, |
| 159 | Message { |
| 160 | role: Role::Assistant, |
| 161 | content: vec![ContentBlock::Text { |
| 162 | text: "initial response".to_string(), |
| 163 | cache_control: None, |
| 164 | }], |
| 165 | }, |
| 166 | ]; |
| 167 | handle |
| 168 | .send(Op::SyncSession { |
| 169 | session_id: Some("edit-mode-test".to_string()), |
| 170 | messages: seeded_messages, |
| 171 | system_prompt: None, |
| 172 | system_prompt_override: false, |
| 173 | model: "deepseek-v4-pro".to_string(), |
| 174 | workspace: tmp.path().to_path_buf(), |
| 175 | mode: AppMode::Agent, |
| 176 | }) |
| 177 | .await |
| 178 | .expect("sync session"); |
| 179 | handle |
| 180 | .send(Op::ChangeMode { |
| 181 | mode: AppMode::Plan, |
| 182 | allow_shell: false, |
| 183 | trust_mode: false, |
| 184 | auto_approve: false, |
| 185 | approval_mode: ApprovalMode::Suggest, |
| 186 | configured_sandbox_mode: None, |
| 187 | }) |
| 188 | .await |
| 189 | .expect("send plan mode"); |
| 190 | handle |
| 191 | .send(Op::EditLastTurn { |
| 192 | new_message: "revise this in plan mode".to_string(), |
| 193 | submission_id: Some("sub-edit-1".to_string()), |
| 194 | }) |
| 195 | .await |
| 196 | .expect("send edit"); |
| 197 | // The replacement turn the edit replays must echo the edit's own |
| 198 | // correlation token, not the one of any earlier turn. |
| 199 | { |
| 200 | let mut rx = handle.rx_event.write().await; |
| 201 | loop { |
| 202 | let event = tokio::time::timeout(model_turn_event_timeout(), rx.recv()) |
| 203 | .await |
| 204 | .expect("timed out waiting for the edited turn") |
| 205 | .expect("engine event"); |
| 206 | if let Event::TurnStarted { submission_id, .. } = event { |
| 207 | assert_eq!( |
| 208 | submission_id.as_deref(), |
| 209 | Some("sub-edit-1"), |
| 210 | "the edit's replacement turn must echo its correlation token" |
| 211 | ); |
| 212 | break; |
| 213 | } |
| 214 | } |
| 215 | } |
| 216 | |
| 217 | let (tx, rx) = tokio::sync::oneshot::channel(); |
| 218 | handle |
| 219 | .send(Op::GetSessionSnapshot { |
| 220 | tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))), |
| 221 | }) |
| 222 | .await |
| 223 | .expect("request snapshot"); |
| 224 | let snapshot = tokio::time::timeout(model_turn_event_timeout(), rx) |
| 225 | .await |
| 226 | .expect("snapshot response") |
| 227 | .expect("snapshot"); |
| 228 | |
| 229 | assert_eq!(snapshot.mode, "plan"); |
| 230 | |
| 231 | let requests = server |
| 232 | .received_requests() |
| 233 | .await |
| 234 | .expect("recorded replacement request"); |
| 235 | assert_eq!( |
| 236 | requests.len(), |
| 237 | 1, |
| 238 | "edit must dispatch exactly one replacement turn" |
| 239 | ); |
| 240 | handle.send(Op::Shutdown).await.expect("shutdown engine"); |
| 241 | run.await.expect("engine task"); |
| 242 | } |
| 243 | |
| 244 | #[tokio::test] |
| 245 | #[allow(clippy::await_holding_lock)] |
| 246 | async fn edit_last_turn_cuts_at_user_prompt_before_tool_results() { |
| 247 | use wiremock::matchers::{method, path}; |
| 248 | use wiremock::{Mock, MockServer, ResponseTemplate}; |
| 249 | |
| 250 | // Tool results persist with role "user"; the edit cut must land on the |
| 251 | // last genuine user prompt, not on the trailing tool_result of the |
| 252 | // previous turn. |
| 253 | let _lock = lock_test_env(); |
| 254 | let tmp = tempdir().expect("tempdir"); |
| 255 | let server = MockServer::start().await; |
| 256 | let done_sse = concat!( |
| 257 | "data: {\"id\":\"chatcmpl-edit-cut\",\"choices\":[{\"index\":0,", |
| 258 | "\"delta\":{\"content\":\"Revised answer.\"},\"finish_reason\":null}]}\n\n", |
| 259 | "data: {\"id\":\"chatcmpl-edit-cut\",\"choices\":[{\"index\":0,", |
| 260 | "\"delta\":{},\"finish_reason\":\"stop\"}]}\n\n", |
| 261 | "data: [DONE]\n\n", |
| 262 | ); |
| 263 | Mock::given(method("POST")) |
| 264 | .and(path("/v1/chat/completions")) |
| 265 | .respond_with( |
| 266 | ResponseTemplate::new(200) |
| 267 | .insert_header("content-type", "text/event-stream") |
| 268 | .set_body_string(done_sse), |
| 269 | ) |
| 270 | .expect(1) |
| 271 | .mount(&server) |
| 272 | .await; |
| 273 | |
| 274 | let api_config = Config { |
| 275 | ..Config::default() |
| 276 | } |
| 277 | .with_legacy_root(Some("test-key".to_string()), Some(server.uri())); |
| 278 | let config = EngineConfig { |
| 279 | workspace: tmp.path().to_path_buf(), |
| 280 | model: "deepseek-v4-pro".to_string(), |
| 281 | snapshots_enabled: false, |
| 282 | subagents_enabled: false, |
| 283 | ..Default::default() |
| 284 | }; |
| 285 | let (engine, handle) = Engine::new(config, &api_config); |
| 286 | |
| 287 | let run = tokio::spawn(engine.run()); |
| 288 | let seeded_messages = vec![ |
| 289 | Message { |
| 290 | role: Role::User, |
| 291 | content: vec![ContentBlock::Text { |
| 292 | text: "original prompt".to_string(), |
| 293 | cache_control: None, |
| 294 | }], |
| 295 | }, |
| 296 | Message { |
| 297 | role: Role::Assistant, |
| 298 | content: vec![ContentBlock::ToolUse { |
| 299 | execution_id: None, |
| 300 | id: "call_1".to_string(), |
| 301 | name: "Bash".to_string(), |
| 302 | input: serde_json::json!({"command": "printf hi"}), |
| 303 | caller: None, |
| 304 | thought_signature: None, |
| 305 | }], |
| 306 | }, |
| 307 | Message { |
| 308 | role: Role::User, |
| 309 | content: vec![ContentBlock::ToolResult { |
| 310 | execution_id: None, |
| 311 | tool_use_id: "call_1".to_string(), |
| 312 | content: "unique-tool-output-marker".to_string(), |
| 313 | is_error: None, |
| 314 | content_blocks: None, |
| 315 | }], |
| 316 | }, |
| 317 | Message { |
| 318 | role: Role::Assistant, |
| 319 | content: vec![ContentBlock::Text { |
| 320 | text: "final answer".to_string(), |
| 321 | cache_control: None, |
| 322 | }], |
| 323 | }, |
| 324 | ]; |
| 325 | handle |
| 326 | .send(Op::SyncSession { |
| 327 | session_id: Some("edit-cut-test".to_string()), |
| 328 | messages: seeded_messages, |
| 329 | system_prompt: None, |
| 330 | system_prompt_override: false, |
| 331 | model: "deepseek-v4-pro".to_string(), |
| 332 | workspace: tmp.path().to_path_buf(), |
| 333 | mode: AppMode::Agent, |
| 334 | }) |
| 335 | .await |
| 336 | .expect("sync session"); |
| 337 | handle |
| 338 | .send(Op::EditLastTurn { |
| 339 | new_message: "edited prompt".to_string(), |
| 340 | submission_id: None, |
| 341 | }) |
| 342 | .await |
| 343 | .expect("send edit"); |
| 344 | |
| 345 | // Ops are processed in order: once the snapshot arrives, the replacement |
| 346 | // turn has completed. |
| 347 | let (tx, rx) = tokio::sync::oneshot::channel(); |
| 348 | handle |
| 349 | .send(Op::GetSessionSnapshot { |
| 350 | tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))), |
| 351 | }) |
| 352 | .await |
| 353 | .expect("request snapshot"); |
| 354 | let snapshot = tokio::time::timeout(model_turn_event_timeout(), rx) |
| 355 | .await |
| 356 | .expect("snapshot response") |
| 357 | .expect("snapshot"); |
| 358 | |
| 359 | assert_eq!( |
| 360 | snapshot.messages.len(), |
| 361 | 2, |
| 362 | "the whole previous turn (prompt, tool_use, tool_result, answer) must be cut: {:?}", |
| 363 | snapshot.messages |
| 364 | ); |
| 365 | assert!( |
| 366 | snapshot.messages.iter().all(|message| !message |
| 367 | .content |
| 368 | .iter() |
| 369 | .any(|block| matches!(block, ContentBlock::ToolResult { .. }))), |
| 370 | "no tool_result may survive the cut: {:?}", |
| 371 | snapshot.messages |
| 372 | ); |
| 373 | let replacement_text = message_text_of(&snapshot.messages[0]); |
| 374 | assert!( |
| 375 | replacement_text.contains("edited prompt"), |
| 376 | "first surviving message is the edited prompt: {replacement_text}" |
| 377 | ); |
| 378 | |
| 379 | let requests = server |
| 380 | .received_requests() |
| 381 | .await |
| 382 | .expect("recorded replacement request"); |
| 383 | assert_eq!(requests.len(), 1); |
| 384 | let body = String::from_utf8(requests[0].body.clone()).expect("request body utf8"); |
| 385 | assert!(body.contains("edited prompt")); |
| 386 | assert!( |
| 387 | !body.contains("original prompt"), |
| 388 | "old prompt must not leak into the replacement turn: {body}" |
| 389 | ); |
| 390 | assert!( |
| 391 | !body.contains("unique-tool-output-marker"), |
| 392 | "tool round-trip must not leak into the replacement turn: {body}" |
| 393 | ); |
| 394 | |
| 395 | handle.send(Op::Shutdown).await.expect("shutdown engine"); |
| 396 | run.await.expect("engine task"); |
| 397 | } |
| 398 | |
| 399 | #[tokio::test] |
| 400 | #[allow(clippy::await_holding_lock)] |
| 401 | async fn edit_last_turn_without_user_prompt_errors_and_sends_nothing() { |
| 402 | use wiremock::matchers::{method, path}; |
| 403 | use wiremock::{Mock, MockServer, ResponseTemplate}; |
| 404 | |
| 405 | let _lock = lock_test_env(); |
| 406 | let tmp = tempdir().expect("tempdir"); |
| 407 | let server = MockServer::start().await; |
| 408 | Mock::given(method("POST")) |
| 409 | .and(path("/v1/chat/completions")) |
| 410 | .respond_with(ResponseTemplate::new(200)) |
| 411 | .expect(0) |
| 412 | .mount(&server) |
| 413 | .await; |
| 414 | |
| 415 | let api_config = Config { |
| 416 | ..Config::default() |
| 417 | } |
| 418 | .with_legacy_root(Some("test-key".to_string()), Some(server.uri())); |
| 419 | let config = EngineConfig { |
| 420 | workspace: tmp.path().to_path_buf(), |
| 421 | model: "deepseek-v4-pro".to_string(), |
| 422 | snapshots_enabled: false, |
| 423 | subagents_enabled: false, |
| 424 | ..Default::default() |
| 425 | }; |
| 426 | let (engine, handle) = Engine::new(config, &api_config); |
| 427 | |
| 428 | let run = tokio::spawn(engine.run()); |
| 429 | // History without any genuine user prompt: nothing to edit. The engine |
| 430 | // must surface an error instead of silently appending the message. |
| 431 | handle |
| 432 | .send(Op::SyncSession { |
| 433 | session_id: Some("edit-no-user-test".to_string()), |
| 434 | messages: vec![Message { |
| 435 | role: Role::Assistant, |
| 436 | content: vec![ContentBlock::Text { |
| 437 | text: "assistant only".to_string(), |
| 438 | cache_control: None, |
| 439 | }], |
| 440 | }], |
| 441 | system_prompt: None, |
| 442 | system_prompt_override: false, |
| 443 | model: "deepseek-v4-pro".to_string(), |
| 444 | workspace: tmp.path().to_path_buf(), |
| 445 | mode: AppMode::Agent, |
| 446 | }) |
| 447 | .await |
| 448 | .expect("sync session"); |
| 449 | handle |
| 450 | .send(Op::EditLastTurn { |
| 451 | new_message: "edited prompt".to_string(), |
| 452 | submission_id: None, |
| 453 | }) |
| 454 | .await |
| 455 | .expect("send edit"); |
| 456 | |
| 457 | let deadline = tokio::time::Instant::now() + Duration::from_secs(5); |
| 458 | let mut saw_edit_error = false; |
| 459 | let mut saw_failed_terminal = false; |
| 460 | { |
| 461 | let mut events = handle.rx_event.write().await; |
| 462 | while let Ok(Some(event)) = tokio::time::timeout_at(deadline, events.recv()).await { |
| 463 | match event { |
| 464 | Event::Error { envelope, .. } => { |
| 465 | assert_eq!(envelope.code, "edit_last_turn_no_user_prompt"); |
| 466 | assert!(!envelope.recoverable); |
| 467 | assert!( |
| 468 | envelope.message.contains("no user message"), |
| 469 | "unexpected error: {}", |
| 470 | envelope.message |
| 471 | ); |
| 472 | saw_edit_error = true; |
| 473 | } |
| 474 | Event::TurnComplete { status, error, .. } => { |
| 475 | assert_eq!(status, TurnOutcomeStatus::Failed); |
| 476 | assert!( |
| 477 | error |
| 478 | .as_deref() |
| 479 | .is_some_and(|message| message.contains("no user message")), |
| 480 | "failed edit terminal must carry the rejection: {error:?}" |
| 481 | ); |
| 482 | saw_failed_terminal = true; |
| 483 | break; |
| 484 | } |
| 485 | _ => {} |
| 486 | } |
| 487 | } |
| 488 | } |
| 489 | assert!(saw_edit_error, "edit without a user prompt must error out"); |
| 490 | assert!( |
| 491 | saw_failed_terminal, |
| 492 | "edit rejection must complete the submitted host lifecycle" |
| 493 | ); |
| 494 | |
| 495 | let (tx, rx) = tokio::sync::oneshot::channel(); |
| 496 | handle |
| 497 | .send(Op::GetSessionSnapshot { |
| 498 | tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))), |
| 499 | }) |
| 500 | .await |
| 501 | .expect("request snapshot"); |
| 502 | let snapshot = tokio::time::timeout(Duration::from_secs(2), rx) |
| 503 | .await |
| 504 | .expect("snapshot response") |
| 505 | .expect("snapshot"); |
| 506 | assert_eq!( |
| 507 | snapshot.messages.len(), |
| 508 | 1, |
| 509 | "failed edit must not append the new message: {:?}", |
| 510 | snapshot.messages |
| 511 | ); |
| 512 | |
| 513 | // An unsupported latest user turn is still a history boundary. It must |
| 514 | // fail in place rather than falling through to the older text prompt and |
| 515 | // deleting a larger portion of the conversation. |
| 516 | let image_only_history = vec![ |
| 517 | Message { |
| 518 | role: Role::User, |
| 519 | content: vec![ContentBlock::Text { |
| 520 | text: "older editable prompt".to_string(), |
| 521 | cache_control: None, |
| 522 | }], |
| 523 | }, |
| 524 | Message { |
| 525 | role: Role::Assistant, |
| 526 | content: vec![ContentBlock::Text { |
| 527 | text: "older response".to_string(), |
| 528 | cache_control: None, |
| 529 | }], |
| 530 | }, |
| 531 | Message { |
| 532 | role: Role::User, |
| 533 | content: vec![ContentBlock::ImageUrl { |
| 534 | image_url: codewhale_models::ImageUrlContent { |
| 535 | url: "data:image/png;base64,AAAA".to_string(), |
| 536 | }, |
| 537 | }], |
| 538 | }, |
| 539 | ]; |
| 540 | handle |
| 541 | .send(Op::SyncSession { |
| 542 | session_id: Some("edit-unsupported-user-test".to_string()), |
| 543 | messages: image_only_history.clone(), |
| 544 | system_prompt: None, |
| 545 | system_prompt_override: false, |
| 546 | model: "deepseek-v4-pro".to_string(), |
| 547 | workspace: tmp.path().to_path_buf(), |
| 548 | mode: AppMode::Agent, |
| 549 | }) |
| 550 | .await |
| 551 | .expect("sync unsupported user session"); |
| 552 | handle |
| 553 | .send(Op::EditLastTurn { |
| 554 | new_message: "must not replace the older prompt".to_string(), |
| 555 | submission_id: None, |
| 556 | }) |
| 557 | .await |
| 558 | .expect("send unsupported edit"); |
| 559 | |
| 560 | let deadline = tokio::time::Instant::now() + Duration::from_secs(5); |
| 561 | let mut saw_unsupported_error = false; |
| 562 | let mut saw_unsupported_terminal = false; |
| 563 | { |
| 564 | let mut events = handle.rx_event.write().await; |
| 565 | while let Ok(Some(event)) = tokio::time::timeout_at(deadline, events.recv()).await { |
| 566 | match event { |
| 567 | Event::Error { envelope, .. } => { |
| 568 | assert_eq!(envelope.code, "edit_last_turn_unsupported_user_content"); |
| 569 | assert!(!envelope.recoverable); |
| 570 | saw_unsupported_error = true; |
| 571 | } |
| 572 | Event::TurnComplete { status, error, .. } => { |
| 573 | assert_eq!(status, TurnOutcomeStatus::Failed); |
| 574 | assert!( |
| 575 | error |
| 576 | .as_deref() |
| 577 | .is_some_and(|message| message |
| 578 | .contains("latest user message has no editable text")), |
| 579 | "unsupported edit terminal must carry the rejection: {error:?}" |
| 580 | ); |
| 581 | saw_unsupported_terminal = true; |
| 582 | break; |
| 583 | } |
| 584 | _ => {} |
| 585 | } |
| 586 | } |
| 587 | } |
| 588 | assert!(saw_unsupported_error); |
| 589 | assert!(saw_unsupported_terminal); |
| 590 | |
| 591 | let (tx, rx) = tokio::sync::oneshot::channel(); |
| 592 | handle |
| 593 | .send(Op::GetSessionSnapshot { |
| 594 | tx: std::sync::Arc::new(std::sync::Mutex::new(Some(tx))), |
| 595 | }) |
| 596 | .await |
| 597 | .expect("request unsupported snapshot"); |
| 598 | let unsupported_snapshot = tokio::time::timeout(Duration::from_secs(2), rx) |
| 599 | .await |
| 600 | .expect("unsupported snapshot response") |
| 601 | .expect("unsupported snapshot"); |
| 602 | assert_eq!( |
| 603 | unsupported_snapshot.messages, image_only_history, |
| 604 | "unsupported latest user content must leave the entire history unchanged" |
| 605 | ); |
| 606 | |
| 607 | let requests = server.received_requests().await.expect("recorded requests"); |
| 608 | assert!( |
| 609 | requests.is_empty(), |
| 610 | "failed edit must not dispatch a provider turn" |
| 611 | ); |
| 612 | |
| 613 | handle.send(Op::Shutdown).await.expect("shutdown engine"); |
| 614 | run.await.expect("engine task"); |
| 615 | } |
| 616 | |
| 617 | #[tokio::test] |
| 618 | async fn provider_runtime_status_reports_configured_zai_cap_without_client() { |
| 619 | let (engine, handle) = { |
| 620 | let _lock = lock_test_env(); |
| 621 | let _zai_key = EnvVarGuard::remove("ZAI_API_KEY"); |
| 622 | let _zai_alt_key = EnvVarGuard::remove("Z_AI_API_KEY"); |
| 623 | let api_config = Config { |
| 624 | provider: Some("zai".to_string()), |
| 625 | ..Config::default() |
| 626 | }; |
| 627 | Engine::new(EngineConfig::default(), &api_config) |
| 628 | }; |
| 629 | |
| 630 | let run = tokio::spawn(engine.run()); |
| 631 | let status = tokio::time::timeout(Duration::from_secs(2), handle.get_provider_runtime_status()) |
| 632 | .await |
| 633 | .expect("provider runtime status response") |
| 634 | .expect("provider runtime status"); |
| 635 | |
| 636 | assert_eq!(status.provider, ProviderKind::Zai); |
| 637 | assert_eq!( |
| 638 | status.request_concurrency_limit, |
| 639 | Some(crate::config::DEFAULT_ZAI_PROVIDER_MAX_CONCURRENCY) |
| 640 | ); |
| 641 | assert_eq!(status.active_provider_requests, 0); |
| 642 | |
| 643 | run.abort(); |
| 644 | } |
| 645 | |
| 646 | #[test] |
| 647 | fn detects_context_length_errors_from_provider_payloads() { |
| 648 | let msg = r#"SSE stream request failed: HTTP 400 Bad Request: {"error":{"message":"This model's maximum context length is 131072 tokens. However, you requested 153056 tokens (148960 in the messages, 4096 in the completion).","type":"invalid_request_error"}}"#; |
| 649 | assert!(is_context_length_error_message(msg)); |
| 650 | // llama.cpp's server wording (#6374): a genuine overflow on a local route |
| 651 | // must enter the bounded recovery path too. |
| 652 | assert!(is_context_length_error_message( |
| 653 | r#"SSE stream request failed: HTTP 400 Bad Request: {"error":{"code":400,"message":"the request exceeds the available context size. try increasing the context size or enable context shift","type":"invalid_request_error"}}"# |
| 654 | )); |
| 655 | assert!(!is_context_length_error_message( |
| 656 | "SSE stream request failed: HTTP 400 Bad Request: model not found" |
| 657 | )); |
| 658 | } |
| 659 | |
| 660 | /// #6374: the exhausted-recovery message must name levers the reader has. |
| 661 | #[test] |
| 662 | fn context_overflow_exhausted_message_names_levers_that_exist_in_the_mode() { |
| 663 | let headless = super::context::context_overflow_exhausted_message(false, 2, 98_739, 97_280); |
| 664 | assert!( |
| 665 | !headless.contains("/compact") && !headless.contains("/clear"), |
| 666 | "a headless host has no command layer: {headless}" |
| 667 | ); |
| 668 | assert!( |
| 669 | headless.contains("2 emergency compaction passes"), |
| 670 | "{headless}" |
| 671 | ); |
| 672 | assert!( |
| 673 | headless.contains("CODEWHALE_MAX_OUTPUT_TOKENS"), |
| 674 | "{headless}" |
| 675 | ); |
| 676 | let interactive = super::context::context_overflow_exhausted_message(true, 1, 98_739, 97_280); |
| 677 | assert!( |
| 678 | interactive.contains("/compact") && interactive.contains("/clear"), |
| 679 | "{interactive}" |
| 680 | ); |
| 681 | assert!( |
| 682 | interactive.contains("1 emergency compaction pass "), |
| 683 | "{interactive}" |
| 684 | ); |
| 685 | } |
| 686 | |
| 687 | #[test] |
| 688 | fn context_budget_scenario() { |
| 689 | // Scenario consolidation of: context_budget_reserves_output_and_headroom, context_budget_uses_conservative_fallback_for_unknown_models, context_budget_uses_provider_effective_window_for_openai_codex |
| 690 | // from context_budget_reserves_output_and_headroom |
| 691 | { |
| 692 | // Serialize with other tests that mutate DEEPSEEK_MAX_OUTPUT_TOKENS so |
| 693 | // the internal effective_max_output_tokens() call sees a stable env. |
| 694 | let _lock = lock_test_env(); |
| 695 | // Preflight reserves exactly the route-effective output request plus the |
| 696 | // shared safety headroom, even on a 1M route. |
| 697 | let budget = context_input_budget_for_provider(ProviderKind::Deepseek, "deepseek-v4-pro") |
| 698 | .expect("deepseek-v4-pro should have a known context window"); |
| 699 | let v4_window: usize = 1_000_000; |
| 700 | let expected = v4_window |
| 701 | - effective_max_output_tokens_for_route(ProviderKind::Deepseek, "deepseek-v4-pro", None) |
| 702 | as usize |
| 703 | - 1_024usize; |
| 704 | assert_eq!(budget, expected); |
| 705 | } |
| 706 | // from context_budget_uses_conservative_fallback_for_unknown_models |
| 707 | { |
| 708 | let _lock = lock_test_env(); |
| 709 | let budget = context_input_budget_for_provider(ProviderKind::Openai, "auto") |
| 710 | .expect("unknown/auto model ids should still get a conservative hard preflight budget"); |
| 711 | let expected = 128_000usize |
| 712 | - effective_max_output_tokens_for_route(ProviderKind::Openai, "auto", None) as usize |
| 713 | - 1_024usize; |
| 714 | assert_eq!(budget, expected); |
| 715 | } |
| 716 | // from context_budget_uses_provider_effective_window_for_openai_codex |
| 717 | { |
| 718 | let _lock = lock_test_env(); |
| 719 | let budget = context_input_budget_for_provider(ProviderKind::OpenaiCodex, "gpt-5.5") |
| 720 | .expect("OpenAI Codex should use a conservative fallback without route metadata"); |
| 721 | let expected = usize::try_from(crate::config::OPENAI_CODEX_EFFECTIVE_CONTEXT_WINDOW_TOKENS) |
| 722 | .expect("context window fits usize") |
| 723 | - crate::config::provider_capability(ProviderKind::OpenaiCodex, "gpt-5.5") |
| 724 | .max_output |
| 725 | .expect("Codex route publishes a deliberate conservative output cap") |
| 726 | as usize |
| 727 | - 1_024usize; |
| 728 | assert_eq!(budget, expected); |
| 729 | } |
| 730 | } |
| 731 | |
| 732 | #[test] |
| 733 | fn route_context_scenario() { |
| 734 | // Scenario consolidation of: route_context_budget_uses_shared_budget_service, route_context_budget_prefers_resolved_route_limits |
| 735 | // from route_context_budget_uses_shared_budget_service |
| 736 | { |
| 737 | let _lock = lock_test_env(); |
| 738 | let budget = |
| 739 | route_context_budget_for_provider(ProviderKind::OpenaiCodex, "gpt-5.5", 380_000) |
| 740 | .expect("OpenAI Codex should produce a route budget"); |
| 741 | |
| 742 | assert_eq!( |
| 743 | budget.window_tokens, |
| 744 | u64::from(crate::config::OPENAI_CODEX_EFFECTIVE_CONTEXT_WINDOW_TOKENS) |
| 745 | ); |
| 746 | assert_eq!( |
| 747 | budget.output_cap_tokens, |
| 748 | u64::from( |
| 749 | crate::config::provider_capability(ProviderKind::OpenaiCodex, "gpt-5.5") |
| 750 | .max_output |
| 751 | .expect("Codex route publishes a deliberate conservative output cap") |
| 752 | ) |
| 753 | ); |
| 754 | assert_eq!( |
| 755 | budget.pressure, |
| 756 | crate::context_budget::PressureLevel::Critical |
| 757 | ); |
| 758 | assert!(!budget.fits_additional(1)); |
| 759 | } |
| 760 | // from route_context_budget_prefers_resolved_route_limits |
| 761 | { |
| 762 | let _lock = lock_test_env(); |
| 763 | let limits = codewhale_config::route::RouteLimits { |
| 764 | context_tokens: Some(128_000), |
| 765 | input_tokens: None, |
| 766 | output_tokens: Some(32_768), |
| 767 | }; |
| 768 | let budget = route_context_budget_for_route( |
| 769 | ProviderKind::Openrouter, |
| 770 | "deepseek/deepseek-v4-pro", |
| 771 | Some(limits), |
| 772 | 60_000, |
| 773 | ) |
| 774 | .expect("route limits should produce a budget"); |
| 775 | |
| 776 | assert_eq!(budget.window_tokens, 128_000); |
| 777 | assert_eq!(budget.output_cap_tokens, 32_768); |
| 778 | assert_eq!(budget.available_input_tokens, 34_208); |
| 779 | } |
| 780 | } |
| 781 | |
| 782 | #[test] |
| 783 | fn route_input_limit_blocks_oversized_preflight_before_transport() { |
| 784 | let _lock = lock_test_env(); |
| 785 | let limits = codewhale_config::route::RouteLimits { |
| 786 | context_tokens: Some(1_000_000), |
| 787 | input_tokens: Some(128_000), |
| 788 | output_tokens: Some(64_000), |
| 789 | }; |
| 790 | let estimated_input = 200_000; |
| 791 | let budget = route_context_budget_for_route( |
| 792 | ProviderKind::Vllm, |
| 793 | "DeepSeek-V4-Flash", |
| 794 | Some(limits), |
| 795 | estimated_input, |
| 796 | ) |
| 797 | .expect("resolved route limits should produce the turn-loop preflight budget"); |
| 798 | |
| 799 | assert_eq!(budget.window_tokens, 1_000_000); |
| 800 | assert_eq!(budget.output_cap_tokens, 64_000); |
| 801 | assert_eq!(budget.input_budget_ceiling, 128_000); |
| 802 | assert_eq!(budget.available_input_tokens, 0); |
| 803 | assert!( |
| 804 | estimated_input > usize::try_from(budget.input_budget_ceiling).unwrap(), |
| 805 | "the turn-loop preflight must recover before constructing a network request" |
| 806 | ); |
| 807 | } |
| 808 | |
| 809 | /// #6374: the preflight guard measured a ×1.5-inflated estimate against the |
| 810 | /// honest input ceiling, so a route refused at two thirds of its budget with |
| 811 | /// the request never leaving the machine. The window here is calibrated so the |
| 812 | /// honest estimate sits below the ceiling and the inflated one above it; the |
| 813 | /// turn must reach the model with its history untouched. |
| 814 | #[tokio::test] |
| 815 | async fn preflight_guard_measures_honest_input_against_the_input_ceiling() { |
| 816 | let _lock = lock_test_env(); |
| 817 | let _output_env = ScopedDeepSeekMaxOutputTokens::unset(); |
| 818 | let workspace = tempdir().expect("workspace"); |
| 819 | let _home = EnvVarGuard::set("CODEWHALE_HOME", workspace.path()); |
| 820 | let mock = std::sync::Arc::new(crate::llm_client::mock::MockLlmClient::new(vec![ |
| 821 | crate::llm_client::mock::canned::simple_text_turn("continuing"), |
| 822 | ])); |
| 823 | let (mut engine, _handle) = Engine::new_with_model_client( |
| 824 | EngineConfig { |
| 825 | terminal_chrome_enabled: false, |
| 826 | ..deterministic_engine_config(workspace.path()) |
| 827 | }, |
| 828 | &Config::default(), |
| 829 | mock.clone(), |
| 830 | ); |
| 831 | // Only the preflight guard is under test; the auto-compaction gate stays out. |
| 832 | engine.config.compaction.enabled = false; |
| 833 | let history: Vec<Message> = [ |
| 834 | (Role::User, "x".repeat(120_000)), |
| 835 | (Role::Assistant, "y".repeat(100_000)), |
| 836 | (Role::User, "please continue".to_string()), |
| 837 | ] |
| 838 | .into_iter() |
| 839 | .map(|(role, text)| Message { |
| 840 | role, |
| 841 | content: vec![ContentBlock::Text { |
| 842 | text, |
| 843 | cache_control: None, |
| 844 | }], |
| 845 | }) |
| 846 | .collect(); |
| 847 | for message in &history { |
| 848 | engine.session.add_message(message.clone()); |
| 849 | } |
| 850 | let system = engine.session.system_prompt.clone(); |
| 851 | let honest = crate::compaction::estimate_input_tokens_for_pressure(&history, system.as_ref()); |
| 852 | let inflated = crate::compaction::estimate_input_tokens_conservative(&history, system.as_ref()); |
| 853 | assert!( |
| 854 | inflated > honest + 20_000, |
| 855 | "fixture must separate the estimators: honest {honest}, inflated {inflated}" |
| 856 | ); |
| 857 | let output_cap = 4_096u64; |
| 858 | let target_ceiling = u64::try_from((honest + inflated) / 2).unwrap(); |
| 859 | engine.active_route_limits = Some(codewhale_config::route::RouteLimits { |
| 860 | context_tokens: Some( |
| 861 | target_ceiling + output_cap + crate::context_budget::CONTEXT_HEADROOM_TOKENS, |
| 862 | ), |
| 863 | input_tokens: None, |
| 864 | output_tokens: Some(output_cap), |
| 865 | }); |
| 866 | let ceiling = route_context_budget_for_route( |
| 867 | engine.api_provider, |
| 868 | &engine.session.model, |
| 869 | engine.active_route_limits, |
| 870 | 0, |
| 871 | ) |
| 872 | .expect("route limits produce a budget") |
| 873 | .input_budget_ceiling; |
| 874 | let ceiling = usize::try_from(ceiling).unwrap(); |
| 875 | assert!( |
| 876 | honest < ceiling && ceiling < inflated, |
| 877 | "calibration: honest {honest} < ceiling {ceiling} < inflated {inflated}" |
| 878 | ); |
| 879 | |
| 880 | let registry = |
| 881 | crate::tools::ToolRegistry::new(crate::tools::spec::ToolContext::new(workspace.path())); |
| 882 | let catalog = registry.to_api_tools_with_cache(true); |
| 883 | let surface = crate::core::engine::tool_catalog::ToolSurfacePolicy::new( |
| 884 | registry, |
| 885 | Some(catalog), |
| 886 | codewhale_config::AppMode::Agent, |
| 887 | &engine.config.tools_always_load, |
| 888 | &[], |
| 889 | false, |
| 890 | None, |
| 891 | None, |
| 892 | Some(4), |
| 893 | crate::core::engine::tool_catalog::ToolMode::Direct, |
| 894 | ); |
| 895 | let (status, error) = engine |
| 896 | .run_turn( |
| 897 | &mut crate::core::turn::TurnContext::new(8), |
| 898 | surface, |
| 899 | None, |
| 900 | None, |
| 901 | ) |
| 902 | .await; |
| 903 | assert_eq!(status, TurnOutcomeStatus::Completed, "{error:?}"); |
| 904 | assert_eq!( |
| 905 | mock.call_count(), |
| 906 | 1, |
| 907 | "the only model request is the turn itself, not an emergency compaction" |
| 908 | ); |
| 909 | let request = mock.last_request().expect("the turn reached the model"); |
| 910 | assert_eq!( |
| 911 | request.messages.len(), |
| 912 | history.len(), |
| 913 | "history reached the model without an emergency compaction pass" |
| 914 | ); |
| 915 | } |
| 916 | |
| 917 | #[test] |
| 918 | fn kimi_catalog_output_ceiling_does_not_collapse_input_budget() { |
| 919 | let _lock = lock_test_env(); |
| 920 | let _guard = ScopedDeepSeekMaxOutputTokens::unset(); |
| 921 | let documented = |
| 922 | route_context_budget_for_route(ProviderKind::Moonshot, "kimi-k2.7-code", None, 0) |
| 923 | .expect("bundled Kimi limits should produce a budget"); |
| 924 | assert_eq!(documented.window_tokens, 262_144); |
| 925 | assert_eq!(documented.output_cap_tokens, 32_768); |
| 926 | assert_eq!(documented.available_input_tokens, 228_352); |
| 927 | |
| 928 | // #4368/#4378: Models.dev may report Kimi's full 262K context as both its |
| 929 | // context window and provider output ceiling. That ceiling must not be |
| 930 | // reserved as though every normal turn requested 262K of output; the |
| 931 | // integrated Kimi route cap is 32K. |
| 932 | let limits = codewhale_config::route::RouteLimits { |
| 933 | context_tokens: Some(262_144), |
| 934 | input_tokens: None, |
| 935 | output_tokens: Some(262_144), |
| 936 | }; |
| 937 | |
| 938 | let budget = |
| 939 | route_context_budget_for_route(ProviderKind::Moonshot, "kimi-k2.7-code", Some(limits), 0) |
| 940 | .expect("Kimi route limits should produce a budget"); |
| 941 | |
| 942 | assert_eq!(budget.window_tokens, 262_144); |
| 943 | assert_eq!(budget.output_cap_tokens, 32_768); |
| 944 | assert_eq!(budget.available_input_tokens, 228_352); |
| 945 | } |
| 946 | |
| 947 | #[test] |
| 948 | fn effective_max_scenario() { |
| 949 | // Scenario consolidation of: effective_max_output_tokens_for_route_caps_to_route_output_limit, effective_max_output_tokens_for_route_caps_to_context_window, effective_max_output_tokens_for_route_keeps_tiny_window_positive, effective_max_output_tokens_caps_api_request_for_large_window_models, effective_max_output_tokens_env_override_rejects_zero_and_invalid |
| 950 | // from effective_max_output_tokens_for_route_caps_to_route_output_limit |
| 951 | { |
| 952 | let _lock = lock_test_env(); |
| 953 | let limits = codewhale_config::route::RouteLimits { |
| 954 | context_tokens: Some(1_000_000), |
| 955 | input_tokens: None, |
| 956 | output_tokens: Some(8_192), |
| 957 | }; |
| 958 | |
| 959 | assert_eq!( |
| 960 | effective_max_output_tokens_for_route( |
| 961 | ProviderKind::Deepseek, |
| 962 | "deepseek-v4-pro", |
| 963 | Some(limits), |
| 964 | ), |
| 965 | 8_192 |
| 966 | ); |
| 967 | } |
| 968 | // from effective_max_output_tokens_for_route_caps_to_context_window |
| 969 | { |
| 970 | let _lock = lock_test_env(); |
| 971 | let limits = codewhale_config::route::RouteLimits { |
| 972 | context_tokens: Some(32_000), |
| 973 | input_tokens: None, |
| 974 | output_tokens: None, |
| 975 | }; |
| 976 | |
| 977 | let cap = effective_max_output_tokens_for_route( |
| 978 | ProviderKind::Deepseek, |
| 979 | "deepseek-v4-pro", |
| 980 | Some(limits), |
| 981 | ); |
| 982 | |
| 983 | assert!(cap < 32_000, "request cap must fit the configured window"); |
| 984 | assert!( |
| 985 | cap > 0, |
| 986 | "small configured windows should still allow output" |
| 987 | ); |
| 988 | } |
| 989 | // from effective_max_output_tokens_for_route_keeps_tiny_window_positive |
| 990 | { |
| 991 | let _lock = lock_test_env(); |
| 992 | let limits = codewhale_config::route::RouteLimits { |
| 993 | context_tokens: Some(2_048), |
| 994 | input_tokens: None, |
| 995 | output_tokens: None, |
| 996 | }; |
| 997 | |
| 998 | assert_eq!( |
| 999 | effective_max_output_tokens_for_route( |
| 1000 | ProviderKind::Deepseek, |
| 1001 | "deepseek-v4-pro", |
| 1002 | Some(limits), |
| 1003 | ), |
| 1004 | 1 |
| 1005 | ); |
| 1006 | } |
| 1007 | // from effective_max_output_tokens_caps_api_request_for_large_window_models |
| 1008 | { |
| 1009 | // Serialize with other tests that mutate DEEPSEEK_MAX_OUTPUT_TOKENS so |
| 1010 | // v4_cap and flash_cap below see the same env state. |
| 1011 | let _lock = lock_test_env(); |
| 1012 | // Hosted V4 documents a 384K capability ceiling in the bundled catalogue, |
| 1013 | // but a ceiling is not a safe no-config request size. The operator can |
| 1014 | // still request a larger value explicitly; the automatic request starts |
| 1015 | // at the ordinary 64K cap (#5516/#5518). |
| 1016 | let v4_cap = effective_max_output_tokens("deepseek-v4-pro"); |
| 1017 | assert_eq!( |
| 1018 | v4_cap, 65_536, |
| 1019 | "hosted V4 must not turn the 384K capability maximum into the default request, got {v4_cap}" |
| 1020 | ); |
| 1021 | |
| 1022 | let flash_cap = effective_max_output_tokens("deepseek-v4-flash"); |
| 1023 | assert_eq!(v4_cap, flash_cap); |
| 1024 | } |
| 1025 | // from effective_max_output_tokens_env_override_rejects_zero_and_invalid |
| 1026 | { |
| 1027 | let _lock = lock_test_env(); |
| 1028 | // Establish the heuristic baseline with the env unset. |
| 1029 | let baseline = { |
| 1030 | let _guard = ScopedDeepSeekMaxOutputTokens::unset(); |
| 1031 | effective_max_output_tokens("deepseek-v4-pro") |
| 1032 | }; |
| 1033 | assert!(baseline > 0); |
| 1034 | |
| 1035 | // 0, non-numeric, and empty values must all fall through to the heuristic |
| 1036 | // rather than producing a zero/garbage cap that would silently break |
| 1037 | // request budgeting. |
| 1038 | for raw in ["0", "abc", "", " ", "-1"] { |
| 1039 | let _guard = ScopedDeepSeekMaxOutputTokens::set(raw); |
| 1040 | assert_eq!( |
| 1041 | effective_max_output_tokens("deepseek-v4-pro"), |
| 1042 | baseline, |
| 1043 | "env={raw:?} should fall through to heuristic" |
| 1044 | ); |
| 1045 | } |
| 1046 | } |
| 1047 | } |
| 1048 | |
| 1049 | #[test] |
| 1050 | fn codex_route_without_output_metadata_uses_oauth_capability_floor() { |
| 1051 | let _lock = lock_test_env(); |
| 1052 | let limits = codewhale_config::route::RouteLimits { |
| 1053 | context_tokens: Some(272_000), |
| 1054 | input_tokens: None, |
| 1055 | output_tokens: None, |
| 1056 | }; |
| 1057 | |
| 1058 | assert_eq!( |
| 1059 | effective_max_output_tokens_for_route(ProviderKind::OpenaiCodex, "gpt-5.5", Some(limits)), |
| 1060 | 4_096 |
| 1061 | ); |
| 1062 | let budget = |
| 1063 | route_context_budget_for_route(ProviderKind::OpenaiCodex, "gpt-5.5", Some(limits), 0) |
| 1064 | .expect("Codex route budget"); |
| 1065 | assert_eq!(budget.output_cap_tokens, 4_096); |
| 1066 | } |
| 1067 | |
| 1068 | #[test] |
| 1069 | fn reasoning_max_does_not_add_a_second_deepseek_v4_output_reservation() { |
| 1070 | let _lock = lock_test_env(); |
| 1071 | let _codewhale = EnvVarGuard::remove("CODEWHALE_MAX_OUTPUT_TOKENS"); |
| 1072 | let _deepseek = EnvVarGuard::remove("DEEPSEEK_MAX_OUTPUT_TOKENS"); |
| 1073 | let limits = codewhale_config::route::RouteLimits { |
| 1074 | context_tokens: Some(327_680), |
| 1075 | input_tokens: None, |
| 1076 | output_tokens: None, |
| 1077 | }; |
| 1078 | let cap = effective_max_output_tokens_for_route( |
| 1079 | ProviderKind::Vllm, |
| 1080 | "DeepSeek-V4-Flash", |
| 1081 | Some(limits), |
| 1082 | ); |
| 1083 | let request = codewhale_core::request::prepare_primary_turn_request( |
| 1084 | codewhale_core::request::PrimaryTurnRequest { |
| 1085 | model: "DeepSeek-V4-Flash".to_string(), |
| 1086 | messages: Vec::new(), |
| 1087 | max_tokens: cap, |
| 1088 | system: None, |
| 1089 | tools: None, |
| 1090 | tool_choice: None, |
| 1091 | reasoning_effort: Some("max".to_string()), |
| 1092 | }, |
| 1093 | ); |
| 1094 | let budget = route_context_budget_for_route( |
| 1095 | ProviderKind::Vllm, |
| 1096 | "DeepSeek-V4-Flash", |
| 1097 | Some(limits), |
| 1098 | 105_000, |
| 1099 | ) |
| 1100 | .expect("max-reasoning vLLM route budget"); |
| 1101 | |
| 1102 | assert_eq!(request.reasoning_effort.as_deref(), Some("max")); |
| 1103 | assert_eq!(request.max_tokens, 65_536); |
| 1104 | assert_eq!(budget.output_cap_tokens, u64::from(request.max_tokens)); |
| 1105 | assert_eq!(budget.input_budget_ceiling, 261_120); |
| 1106 | assert!(budget.available_input_tokens > 0); |
| 1107 | } |
| 1108 | |
| 1109 | struct ScopedDeepSeekMaxOutputTokens { |
| 1110 | previous: Option<OsString>, |
| 1111 | } |
| 1112 | |
| 1113 | impl ScopedDeepSeekMaxOutputTokens { |
| 1114 | fn set(value: &str) -> Self { |
| 1115 | let previous = std::env::var_os("DEEPSEEK_MAX_OUTPUT_TOKENS"); |
| 1116 | // Safety: tests using this helper serialize with lock_test_env() and |
| 1117 | // restore the original value in Drop. |
| 1118 | unsafe { |
| 1119 | std::env::set_var("DEEPSEEK_MAX_OUTPUT_TOKENS", value); |
| 1120 | } |
| 1121 | Self { previous } |
| 1122 | } |
| 1123 | |
| 1124 | fn unset() -> Self { |
| 1125 | let previous = std::env::var_os("DEEPSEEK_MAX_OUTPUT_TOKENS"); |
| 1126 | // Safety: see set(). |
| 1127 | unsafe { |
| 1128 | std::env::remove_var("DEEPSEEK_MAX_OUTPUT_TOKENS"); |
| 1129 | } |
| 1130 | Self { previous } |
| 1131 | } |
| 1132 | } |
| 1133 | |
| 1134 | impl Drop for ScopedDeepSeekMaxOutputTokens { |
| 1135 | fn drop(&mut self) { |
| 1136 | // Safety: tests using this helper serialize with lock_test_env(). |
| 1137 | unsafe { |
| 1138 | if let Some(previous) = self.previous.take() { |
| 1139 | std::env::set_var("DEEPSEEK_MAX_OUTPUT_TOKENS", previous); |
| 1140 | } else { |
| 1141 | std::env::remove_var("DEEPSEEK_MAX_OUTPUT_TOKENS"); |
| 1142 | } |
| 1143 | } |
| 1144 | } |
| 1145 | } |
| 1146 | |
| 1147 | #[test] |
| 1148 | fn effective_max_output_tokens_env_override_returns_positive_value() { |
| 1149 | let _lock = lock_test_env(); |
| 1150 | let _guard = ScopedDeepSeekMaxOutputTokens::set("16384"); |
| 1151 | |
| 1152 | // Override applies regardless of model — V4 hosted, V4 flash, and |
| 1153 | // self-hosted routes all return the env value verbatim before route clamps. |
| 1154 | assert_eq!(effective_max_output_tokens("deepseek-v4-pro"), 16_384); |
| 1155 | assert_eq!(effective_max_output_tokens("deepseek-v4-flash"), 16_384); |
| 1156 | assert_eq!(effective_max_output_tokens("qwen3-32b-256k"), 16_384); |
| 1157 | } |
| 1158 | |
| 1159 | #[test] |
| 1160 | fn internal_context_budget_uses_the_wire_cap_across_window_sizes() { |
| 1161 | // Serialize with other tests that mutate DEEPSEEK_MAX_OUTPUT_TOKENS so |
| 1162 | // both branches below see a stable env. |
| 1163 | let _lock = lock_test_env(); |
| 1164 | // Large routes use the same effective output cap that reaches the wire. |
| 1165 | let internal_budget = |
| 1166 | context_input_budget_for_provider(ProviderKind::Deepseek, "deepseek-v4-pro") |
| 1167 | .expect("V4 should have a known context window"); |
| 1168 | let v4_window: usize = 1_000_000; |
| 1169 | let expected_internal = v4_window |
| 1170 | - effective_max_output_tokens_for_route(ProviderKind::Deepseek, "deepseek-v4-pro", None) |
| 1171 | as usize |
| 1172 | - 1_024usize; |
| 1173 | assert_eq!(internal_budget, expected_internal); |
| 1174 | |
| 1175 | // A 256K self-hosted deployment uses the same rule and yields a usable |
| 1176 | // positive budget rather than silently disabling preflight/recovery. |
| 1177 | let small_window_budget = |
| 1178 | context_input_budget_for_provider(ProviderKind::Openai, "qwen3-32b-256k") |
| 1179 | .expect("a 256K-suffix model must yield Some budget via the effective-cap branch"); |
| 1180 | let effective_output = |
| 1181 | effective_max_output_tokens_for_route(ProviderKind::Openai, "qwen3-32b-256k", None) |
| 1182 | as usize; |
| 1183 | let expected_small = 256_000 - effective_output - 1_024; |
| 1184 | assert_eq!(small_window_budget, expected_small); |
| 1185 | } |
| 1186 | |
| 1187 | const ROUTE_128K: &str = "deepseek-v3.2-128k"; |
| 1188 | const SESSION_6508: &str = "session-6508"; |
| 1189 | |
| 1190 | fn budget_128k() -> usize { |
| 1191 | crate::route_budget::route_inline_char_budget_for_route( |
| 1192 | ProviderKind::Deepseek, |
| 1193 | ROUTE_128K, |
| 1194 | None, |
| 1195 | ) |
| 1196 | } |
| 1197 | |
| 1198 | fn view_128k(tool_name: &str, output: &ToolResult) -> super::context::ToolResultContextView { |
| 1199 | super::context::tool_result_context_view( |
| 1200 | ProviderKind::Deepseek, |
| 1201 | ROUTE_128K, |
| 1202 | None, |
| 1203 | tool_name, |
| 1204 | output, |
| 1205 | ) |
| 1206 | } |
| 1207 | |
| 1208 | /// Run `f` with the spillover and session-artifact roots under a temp home. |
| 1209 | fn with_artifact_home<R>(f: impl FnOnce(&Path) -> R) -> R { |
| 1210 | let _spill_guard = crate::tools::truncate::TEST_SPILLOVER_GUARD |
| 1211 | .lock() |
| 1212 | .unwrap_or_else(|err| err.into_inner()); |
| 1213 | let home = tempdir().expect("tempdir"); |
| 1214 | let path = home.path().to_path_buf(); |
| 1215 | crate::tools::truncate::with_test_home(&path, || f(&path)) |
| 1216 | } |
| 1217 | |
| 1218 | fn session_artifact_files(home: &Path) -> Vec<PathBuf> { |
| 1219 | let dir = home |
| 1220 | .join(".codewhale") |
| 1221 | .join("sessions") |
| 1222 | .join(SESSION_6508) |
| 1223 | .join("artifacts"); |
| 1224 | fs::read_dir(dir) |
| 1225 | .map(|entries| entries.flatten().map(|entry| entry.path()).collect()) |
| 1226 | .unwrap_or_default() |
| 1227 | } |
| 1228 | |
| 1229 | #[test] |
| 1230 | fn under_budget_results_pass_through_whole_for_every_tool() { |
| 1231 | // #6508: one budget, sized by the route, decides what the model sees. |
| 1232 | // There is no per-tool-name soft limit: a web answer or a shell log that |
| 1233 | // fits the budget reaches the model byte for byte. |
| 1234 | // 3% of a 128K window is 15,360 characters (an operator opt-in can only |
| 1235 | // raise it; see route_budget's tests for the exact values). |
| 1236 | assert!(budget_128k() >= 15_360); |
| 1237 | let content = "w".repeat(14_000); |
| 1238 | let output = ToolResult::success(content.clone()); |
| 1239 | for tool_name in [ |
| 1240 | "exec_shell", |
| 1241 | "web_search", |
| 1242 | "Web", |
| 1243 | "web.run", |
| 1244 | "fetch_url", |
| 1245 | "read_file", |
| 1246 | "run_tests", |
| 1247 | ] { |
| 1248 | let view = view_128k(tool_name, &output); |
| 1249 | assert_eq!(view.text, content, "{tool_name} was cut under the budget"); |
| 1250 | assert!(!view.needs_full_output_artifact); |
| 1251 | } |
| 1252 | } |
| 1253 | |
| 1254 | #[test] |
| 1255 | fn over_budget_result_without_a_saved_copy_says_so_and_asks_for_one() { |
| 1256 | // The view never writes. Without a saved copy it asks the engine for |
| 1257 | // one, and until then it promises no ref it cannot honour. |
| 1258 | let raw = "shell line\n".repeat(4_000); |
| 1259 | let output = ToolResult::success(raw.clone()); |
| 1260 | let view = view_128k("exec_shell", &output); |
| 1261 | |
| 1262 | assert!(view.needs_full_output_artifact); |
| 1263 | assert!(view.text.chars().count() <= budget_128k()); |
| 1264 | assert!(view.text.contains("the full output could not be saved")); |
| 1265 | assert!(view.text.contains("no tool call reaches this copy")); |
| 1266 | assert!(!view.text.contains("retrieve_tool_result")); |
| 1267 | } |
| 1268 | |
| 1269 | #[test] |
| 1270 | fn oversized_tool_output_is_recoverable_before_serial_and_parallel_fanout() { |
| 1271 | use crate::llm_client::mock::{MockLlmClient, canned}; |
| 1272 | use crate::tools::spec::{ToolCapability, ToolSpec}; |
| 1273 | |
| 1274 | struct OutputTool { |
| 1275 | parallel: bool, |
| 1276 | content: String, |
| 1277 | } |
| 1278 | #[async_trait::async_trait] |
| 1279 | impl ToolSpec for OutputTool { |
| 1280 | fn name(&self) -> &str { |
| 1281 | "fixture_output" |
| 1282 | } |
| 1283 | fn description(&self) -> &str { |
| 1284 | "Return output below the spill threshold." |
| 1285 | } |
| 1286 | fn input_schema(&self) -> Value { |
| 1287 | json!({"type": "object"}) |
| 1288 | } |
| 1289 | fn capabilities(&self) -> Vec<ToolCapability> { |
| 1290 | vec![ToolCapability::ReadOnly] |
| 1291 | } |
| 1292 | fn supports_parallel(&self) -> bool { |
| 1293 | self.parallel |
| 1294 | } |
| 1295 | async fn execute(&self, _: Value, _: &ToolContext) -> Result<ToolResult, ToolError> { |
| 1296 | Ok(ToolResult::success(self.content.clone())) |
| 1297 | } |
| 1298 | } |
| 1299 | |
| 1300 | with_artifact_home(|home| { |
| 1301 | tokio::runtime::Builder::new_current_thread() |
| 1302 | .enable_all() |
| 1303 | .build() |
| 1304 | .unwrap() |
| 1305 | .block_on(async { |
| 1306 | let raw = format!("{}MIDDLE{}", "h".repeat(5_000), "t".repeat(5_000)); |
| 1307 | assert!(raw.len() < crate::tools::truncate::SPILLOVER_THRESHOLD_BYTES); |
| 1308 | for (parallel, count) in [(true, 1), (true, 2), (false, 1)] { |
| 1309 | let calls = [ |
| 1310 | ("call-one", "fixture_output", "{}"), |
| 1311 | ("call-two", "fixture_output", "{}"), |
| 1312 | ]; |
| 1313 | let mock = Arc::new(MockLlmClient::new(vec![ |
| 1314 | tool_batch_turn(&calls[..count]), |
| 1315 | canned::simple_text_turn("done"), |
| 1316 | ])); |
| 1317 | let (mut engine, handle) = Engine::new_with_model_client( |
| 1318 | deterministic_engine_config(home), |
| 1319 | &Config::default(), |
| 1320 | mock.clone(), |
| 1321 | ); |
| 1322 | engine.active_route_limits = Some(codewhale_config::route::RouteLimits { |
| 1323 | context_tokens: Some(64_000), |
| 1324 | input_tokens: None, |
| 1325 | output_tokens: Some(4_096), |
| 1326 | }); |
| 1327 | let mut registry = crate::tools::ToolRegistry::new(ToolContext::new(home)); |
| 1328 | registry.register(Arc::new(OutputTool { |
| 1329 | parallel, |
| 1330 | content: raw.clone(), |
| 1331 | })); |
| 1332 | let tools = Some(registry.to_api_tools_with_cache(true)); |
| 1333 | let surface = test_tool_surface(&engine, registry, tools, AppMode::Agent); |
| 1334 | let mut turn = crate::core::turn::TurnContext::new(4); |
| 1335 | let (status, error) = engine.run_turn(&mut turn, surface, None, None).await; |
| 1336 | assert_eq!(status, TurnOutcomeStatus::Completed, "{error:?}"); |
| 1337 | let requests = mock.captured_requests(); |
| 1338 | assert_eq!(requests.len(), 2); |
| 1339 | let mut events = handle.rx_event.write().await; |
| 1340 | let mut completed = 0; |
| 1341 | while let Ok(event) = events.try_recv() { |
| 1342 | let Event::ToolCallComplete { |
| 1343 | result, model_call, .. |
| 1344 | } = event |
| 1345 | else { |
| 1346 | continue; |
| 1347 | }; |
| 1348 | let output = result.expect("tool succeeded"); |
| 1349 | assert_eq!(output.content, raw, "UI keeps the complete output"); |
| 1350 | let metadata = output |
| 1351 | .metadata |
| 1352 | .expect("output must be preserved before fanout"); |
| 1353 | let path = metadata["artifact_path"].as_str().expect("artifact path"); |
| 1354 | assert_eq!(fs::read_to_string(path).unwrap(), raw); |
| 1355 | let reference = metadata["artifact_id"].as_str().unwrap(); |
| 1356 | let results = |
| 1357 | guardian_tool_results(&requests[1], &model_call.unwrap().provider_id); |
| 1358 | assert_eq!(results.len(), 1); |
| 1359 | let (text, _) = results[0]; |
| 1360 | assert!(text.len() <= 7_680, "64K route inline budget"); |
| 1361 | assert!(!text.contains("MIDDLE")); |
| 1362 | assert!(text.contains("retrieve_tool_result")); |
| 1363 | assert!(text.contains(reference), "model and UI share the artifact"); |
| 1364 | completed += 1; |
| 1365 | } |
| 1366 | assert_eq!(completed, count); |
| 1367 | } |
| 1368 | }); |
| 1369 | }); |
| 1370 | } |
| 1371 | |
| 1372 | #[test] |
| 1373 | fn over_budget_result_writes_the_full_output_and_names_its_ref() { |
| 1374 | with_artifact_home(|home| { |
| 1375 | let raw = format!("FIRST LINE\n{}LAST LINE", "shell line\n".repeat(4_000)); |
| 1376 | let mut output = ToolResult::success(raw.clone()); |
| 1377 | assert!(view_128k("exec_shell", &output).needs_full_output_artifact); |
| 1378 | |
| 1379 | assert!( |
| 1380 | crate::tools::truncate::preserve_full_output_for_model_context( |
| 1381 | &mut output, |
| 1382 | "call-over", |
| 1383 | "exec_shell", |
| 1384 | SESSION_6508, |
| 1385 | ) |
| 1386 | ); |
| 1387 | // The UI cell keeps the whole result; only metadata changed. |
| 1388 | assert_eq!(output.content, raw); |
| 1389 | |
| 1390 | let view = view_128k("exec_shell", &output); |
| 1391 | assert!(!view.needs_full_output_artifact); |
| 1392 | assert!(view.text.chars().count() <= budget_128k()); |
| 1393 | assert!(view.text.starts_with("FIRST LINE")); |
| 1394 | assert!(view.text.ends_with("LAST LINE")); |
| 1395 | assert!(view.text.contains("omitted range recovery:")); |
| 1396 | assert!(view.text.contains("ref=\"art_call-over\"")); |
| 1397 | |
| 1398 | let files = session_artifact_files(home); |
| 1399 | assert_eq!(files.len(), 1); |
| 1400 | assert_eq!(fs::read_to_string(&files[0]).expect("artifact"), raw); |
| 1401 | }); |
| 1402 | } |
| 1403 | |
| 1404 | #[test] |
| 1405 | fn spilled_preview_is_refit_not_recut() { |
| 1406 | // Spillover already saved the full output and left a ~40 KB preview. The |
| 1407 | // model view re-fits that preview to the budget with the same ref, and no |
| 1408 | // second artifact (which could only hold the preview) is written. |
| 1409 | with_artifact_home(|home| { |
| 1410 | let raw = format!( |
| 1411 | "HEAD START\n{}TAIL END", |
| 1412 | "spilled output line\n".repeat(16_000) |
| 1413 | ); |
| 1414 | let mut output = ToolResult::success(raw.clone()); |
| 1415 | assert!( |
| 1416 | crate::tools::truncate::apply_spillover_with_artifact( |
| 1417 | &mut output, |
| 1418 | "call-spill", |
| 1419 | "exec_shell", |
| 1420 | SESSION_6508, |
| 1421 | ) |
| 1422 | .is_some() |
| 1423 | ); |
| 1424 | assert_eq!(session_artifact_files(home).len(), 1); |
| 1425 | |
| 1426 | let view = view_128k("exec_shell", &output); |
| 1427 | assert!(!view.needs_full_output_artifact); |
| 1428 | assert!(view.text.chars().count() <= budget_128k()); |
| 1429 | assert!(view.text.starts_with("HEAD START")); |
| 1430 | assert!(view.text.ends_with("TAIL END")); |
| 1431 | assert_eq!( |
| 1432 | view.text.matches("omitted range recovery:").count(), |
| 1433 | 1, |
| 1434 | "exactly one footer: {}", |
| 1435 | &view.text[..200] |
| 1436 | ); |
| 1437 | assert!(view.text.contains("ref=\"art_call-spill\"")); |
| 1438 | assert_eq!(session_artifact_files(home).len(), 1); |
| 1439 | }); |
| 1440 | } |
| 1441 | |
| 1442 | #[test] |
| 1443 | fn legacy_spill_without_ref_says_no_tool_call_reaches_it() { |
| 1444 | let head = "h".repeat(32 * 1024); |
| 1445 | let tail = "t".repeat(8 * 1024); |
| 1446 | let preview = format!("{head}\n\n… footer …\n\n…\n{tail}"); |
| 1447 | let output = ToolResult::success(preview).with_metadata(json!({ |
| 1448 | "spillover_path": "/tmp/tool_outputs/call-legacy.txt", |
| 1449 | "retained_head_bytes": head.len(), |
| 1450 | "retained_tail_bytes": tail.len(), |
| 1451 | "original_byte_count": 300_000, |
| 1452 | "truncated": true |
| 1453 | })); |
| 1454 | |
| 1455 | let view = view_128k("exec_shell", &output); |
| 1456 | assert!(!view.needs_full_output_artifact); |
| 1457 | assert!(view.text.chars().count() <= budget_128k()); |
| 1458 | assert!(view.text.contains("/tmp/tool_outputs/call-legacy.txt")); |
| 1459 | assert!(view.text.contains("no tool call reaches this copy")); |
| 1460 | assert!(!view.text.contains("retrieve_tool_result")); |
| 1461 | } |
| 1462 | |
| 1463 | #[test] |
| 1464 | fn structured_run_tests_summary_is_recoverable_and_leads_with_failures() { |
| 1465 | with_artifact_home(|home| { |
| 1466 | let stdout = format!( |
| 1467 | "{}test result: FAILED. 1 failed", |
| 1468 | "test ok ... ok\n".repeat(3_000) |
| 1469 | ); |
| 1470 | let raw = json!({ |
| 1471 | "success": false, |
| 1472 | "exit_code": 101, |
| 1473 | "stdout": stdout, |
| 1474 | "stderr": "", |
| 1475 | "command": "(cd /repo && cargo test)" |
| 1476 | }) |
| 1477 | .to_string(); |
| 1478 | let mut output = ToolResult::success(raw.clone()).with_metadata(json!({ |
| 1479 | "summary": "1 test failed: tools::git::tests::diff_keeps_the_last_file" |
| 1480 | })); |
| 1481 | assert!(view_128k("run_tests", &output).needs_full_output_artifact); |
| 1482 | assert!( |
| 1483 | crate::tools::truncate::preserve_full_output_for_model_context( |
| 1484 | &mut output, |
| 1485 | "call-tests", |
| 1486 | "run_tests", |
| 1487 | SESSION_6508, |
| 1488 | ) |
| 1489 | ); |
| 1490 | |
| 1491 | let view = view_128k("run_tests", &output); |
| 1492 | assert!(!view.needs_full_output_artifact); |
| 1493 | assert!(view.text.chars().count() <= budget_128k()); |
| 1494 | let failures = view |
| 1495 | .text |
| 1496 | .find("failure summary: 1 test failed: tools::git::tests::diff_keeps_the_last_file") |
| 1497 | .expect("failure summary inline"); |
| 1498 | assert!(failures < view.text.find("stdout:").expect("stdout")); |
| 1499 | assert!(view.text.contains("test result: FAILED")); |
| 1500 | assert!(view.text.contains("ref=\"art_call-tests\"")); |
| 1501 | |
| 1502 | let files = session_artifact_files(home); |
| 1503 | assert_eq!(files.len(), 1); |
| 1504 | assert_eq!(fs::read_to_string(&files[0]).expect("artifact"), raw); |
| 1505 | }); |
| 1506 | } |
| 1507 | |
| 1508 | #[test] |
| 1509 | fn display_compaction_never_writes_artifacts() { |
| 1510 | // The TUI builds its API-message copy with the same pure view. |
| 1511 | with_artifact_home(|home| { |
| 1512 | let output = ToolResult::success("x".repeat(60_000)); |
| 1513 | let text = compact_tool_result_for_route( |
| 1514 | ProviderKind::Deepseek, |
| 1515 | ROUTE_128K, |
| 1516 | None, |
| 1517 | "exec_shell", |
| 1518 | &output, |
| 1519 | ); |
| 1520 | assert!(text.chars().count() <= budget_128k()); |
| 1521 | assert!(session_artifact_files(home).is_empty()); |
| 1522 | }); |
| 1523 | } |
| 1524 | |
| 1525 | #[test] |
| 1526 | fn evidence_bounded_preview_is_not_recompacted() { |
| 1527 | // The adaptive evidence envelope already produced an honest bounded |
| 1528 | // preview (head + footer with the recovery path + tail). The context |
| 1529 | // compactor must pass it through untouched, even beyond the 12K hard |
| 1530 | // limit — re-compacting would strip the recovery contract. |
| 1531 | let content = format!( |
| 1532 | "{}\n\n… 19.0 KiB of output omitted (123 lines) — full output at /tmp/art_call.txt; read it back with the read_file tool or with sed line ranges\n\n…\n{}", |
| 1533 | "h".repeat(32 * 1024), |
| 1534 | "t".repeat(8 * 1024) |
| 1535 | ); |
| 1536 | let output = ToolResult::success(content.clone()).with_metadata(json!({ |
| 1537 | "evidence_available": true, |
| 1538 | "truncated": true, |
| 1539 | "spillover_path": "/tmp/art_call.txt" |
| 1540 | })); |
| 1541 | |
| 1542 | let context = compact_tool_result_for_context("deepseek-v3.2-128k", "Bash", &output); |
| 1543 | assert_eq!(context, content); |
| 1544 | assert!(context.contains("full output at /tmp/art_call.txt")); |
| 1545 | } |
| 1546 | |
| 1547 | #[test] |
| 1548 | fn budgeted_read_result_is_not_truncated_a_second_time_by_the_context_compactor() { |
| 1549 | // C05: `read` bounds itself to an explicit per-call byte budget. The 12K |
| 1550 | // context hard limit used to re-truncate that bounded result into a 900- |
| 1551 | // char snippet, discarding both the content and the continuation footer. |
| 1552 | let content = format!( |
| 1553 | "{}\n\n[Showing lines 1-100 of 2000 (100000-byte output budget). Use offset=101 to continue.]", |
| 1554 | "r".repeat(90_000) |
| 1555 | ); |
| 1556 | let budgeted = ToolResult::success(content.clone()).with_metadata(json!({ |
| 1557 | "evidence_routing": "inline", |
| 1558 | "read_budget_bytes": 100_000 |
| 1559 | })); |
| 1560 | let passed_through = compact_tool_result_for_context("deepseek-v3.2-128k", "read", &budgeted); |
| 1561 | assert_eq!(passed_through, content); |
| 1562 | assert!(passed_through.contains("Use offset=101 to continue")); |
| 1563 | |
| 1564 | // The same bytes without a declared budget still take the ordinary path, |
| 1565 | // which is what proves the metadata (not the tool name) did the work. |
| 1566 | let unbudgeted = ToolResult::success(content.clone()); |
| 1567 | let compacted = compact_tool_result_for_context("deepseek-v3.2-128k", "read", &unbudgeted); |
| 1568 | assert!(compacted.contains(crate::tools::truncate::SPILLOVER_RECOVERY_HINT)); |
| 1569 | assert!(compacted.len() < content.len()); |
| 1570 | |
| 1571 | // A result that overran its own declared budget is not exempt. |
| 1572 | let overrun = ToolResult::success(content).with_metadata(json!({ |
| 1573 | "read_budget_bytes": 1_000 |
| 1574 | })); |
| 1575 | let compacted_overrun = compact_tool_result_for_context("deepseek-v3.2-128k", "read", &overrun); |
| 1576 | assert!(compacted_overrun.contains(crate::tools::truncate::SPILLOVER_RECOVERY_HINT)); |
| 1577 | } |
| 1578 | |
| 1579 | #[test] |
| 1580 | fn codex_tool_retention_uses_oauth_route_window_not_asmall_contract_model_window() { |
| 1581 | let content = "route-effective context\n".repeat(900); |
| 1582 | let output = ToolResult::success(content.clone()); |
| 1583 | let limits = codewhale_config::route::RouteLimits { |
| 1584 | context_tokens: Some(272_000), |
| 1585 | input_tokens: None, |
| 1586 | output_tokens: None, |
| 1587 | }; |
| 1588 | |
| 1589 | // The budget follows the route's 272K window (3% of it, 32,640 |
| 1590 | // characters), so this 21.6K result reaches the model whole. |
| 1591 | assert!( |
| 1592 | crate::route_budget::route_inline_char_budget_for_route( |
| 1593 | ProviderKind::OpenaiCodex, |
| 1594 | "gpt-5.5", |
| 1595 | Some(limits), |
| 1596 | ) >= 32_640 |
| 1597 | ); |
| 1598 | let context = compact_tool_result_for_route( |
| 1599 | ProviderKind::OpenaiCodex, |
| 1600 | "gpt-5.5", |
| 1601 | Some(limits), |
| 1602 | "read_file", |
| 1603 | &output, |
| 1604 | ); |
| 1605 | |
| 1606 | assert_eq!(context, content.trim()); |
| 1607 | } |
| 1608 |