| 1 | #[tokio::test] |
| 2 | async fn trust_warning_is_internal_and_tracks_current_state() { |
| 3 | let _lock = lock_test_env(); |
| 4 | let tmp = tempdir().unwrap(); |
| 5 | let _home = EnvVarGuard::set("CODEWHALE_HOME", tmp.path().join("home")); |
| 6 | let _config = EnvVarGuard::set("CODEWHALE_CONFIG_PATH", tmp.path().join("config.toml")); |
| 7 | fs::create_dir_all(tmp.path().join(".claude/skills")).unwrap(); |
| 8 | let (mut engine, handle) = Engine::new( |
| 9 | EngineConfig { |
| 10 | workspace: tmp.path().to_path_buf(), |
| 11 | ..Default::default() |
| 12 | }, |
| 13 | &Config::default(), |
| 14 | ); |
| 15 | let context = engine.installed_next_turn_prompt_context(); |
| 16 | engine.refresh_pinned_header_for_turn(&context); |
| 17 | let frozen = engine.session.system_prompt.clone(); |
| 18 | let warning = engine |
| 19 | .session |
| 20 | .messages |
| 21 | .last() |
| 22 | .expect("logged warning") |
| 23 | .clone(); |
| 24 | assert_eq!(warning.role, Role::User); |
| 25 | assert!(crate::runtime_handoff::is_internal_runtime_handoff( |
| 26 | &warning |
| 27 | )); |
| 28 | assert!(crate::runtime_handoff::is_runtime_owned_user_message( |
| 29 | &warning |
| 30 | )); |
| 31 | // The transcript side (no history cell for it) is pinned in |
| 32 | // `tui::history::tests`, keeping this runtime test off the UI crate path. |
| 33 | assert!( |
| 34 | crate::compaction::retained_user_messages(std::slice::from_ref(&warning), 4096).is_empty() |
| 35 | ); |
| 36 | let prompt = Message { |
| 37 | role: Role::User, |
| 38 | content: vec![ContentBlock::Text { |
| 39 | text: "Review the code".into(), |
| 40 | cache_control: None, |
| 41 | }], |
| 42 | }; |
| 43 | assert_eq!( |
| 44 | crate::session_manager::conversation_title_prompt(&[warning.clone(), prompt.clone()]), |
| 45 | Some("Review the code") |
| 46 | ); |
| 47 | // Quoting the envelope without engine provenance remains an ordinary user turn. |
| 48 | let quoted = Message { |
| 49 | role: Role::User, |
| 50 | content: vec![warning.content[0].clone()], |
| 51 | }; |
| 52 | assert!(!crate::runtime_handoff::is_internal_runtime_handoff( |
| 53 | "ed |
| 54 | )); |
| 55 | assert!(!crate::runtime_handoff::is_runtime_owned_user_message( |
| 56 | "ed |
| 57 | )); |
| 58 | engine.session.add_message(prompt.clone()); |
| 59 | let initial = engine.session.messages.len(); |
| 60 | engine.refresh_pinned_header_for_turn(&context); |
| 61 | assert_eq!(engine.session.messages.len(), initial); |
| 62 | |
| 63 | crate::config::save_workspace_trust(tmp.path()).unwrap(); |
| 64 | engine.refresh_pinned_header_for_turn(&context); |
| 65 | let resolved = engine.session.messages.last().unwrap().clone(); |
| 66 | assert_ne!(resolved, warning); |
| 67 | assert!(crate::runtime_handoff::is_internal_runtime_handoff( |
| 68 | &resolved |
| 69 | )); |
| 70 | assert!( |
| 71 | matches!(&resolved.content[0], ContentBlock::Text { text, .. } if text.contains("no longer applies")) |
| 72 | ); |
| 73 | engine.refresh_pinned_header_for_turn(&context); |
| 74 | assert_eq!(engine.session.messages.len(), initial + 1); |
| 75 | crate::config::set_workspace_trust(tmp.path(), false).unwrap(); |
| 76 | engine.refresh_pinned_header_for_turn(&context); |
| 77 | assert_eq!(engine.session.messages.last(), Some(&warning)); |
| 78 | assert_eq!( |
| 79 | engine.session.messages.len(), |
| 80 | initial + 2, |
| 81 | "revocation must append a new warning after the correction" |
| 82 | ); |
| 83 | assert_eq!( |
| 84 | engine.session.system_prompt, frozen, |
| 85 | "volatile trust facts never modify the frozen prefix" |
| 86 | ); |
| 87 | assert!( |
| 88 | !codewhale_core::prefix_cache::system_prompt_text(frozen.as_ref()) |
| 89 | .contains("kind=\"workspace_trust\"") |
| 90 | ); |
| 91 | |
| 92 | engine.refresh_system_prompt_with_reason("model"); |
| 93 | engine.emit_session_updated().await; |
| 94 | let event = handle.rx_event.write().await.recv().await.unwrap(); |
| 95 | let Event::SessionUpdated { messages, .. } = event else { |
| 96 | panic!("expected session update"); |
| 97 | }; |
| 98 | assert_eq!( |
| 99 | messages.as_slice(), |
| 100 | &[warning.clone(), prompt, resolved, warning] |
| 101 | ); |
| 102 | } |
| 103 | |
| 104 | fn context_update_messages(engine: &Engine) -> Vec<String> { |
| 105 | engine |
| 106 | .session |
| 107 | .messages |
| 108 | .iter() |
| 109 | .filter(|m| m.role == "user") |
| 110 | .filter_map(|m| match m.content.first() { |
| 111 | Some(ContentBlock::Text { text, .. }) if text.starts_with("<context_update>") => { |
| 112 | Some(text.clone()) |
| 113 | } |
| 114 | _ => None, |
| 115 | }) |
| 116 | .collect() |
| 117 | } |
| 118 | |
| 119 | #[test] |
| 120 | fn workspace_drift_arrives_as_one_context_update_and_never_moves_the_header() { |
| 121 | let _lock = lock_test_env(); |
| 122 | let tmp = tempdir().expect("tempdir"); |
| 123 | let config = EngineConfig { |
| 124 | workspace: tmp.path().to_path_buf(), |
| 125 | project_context_pack_enabled: true, |
| 126 | ..Default::default() |
| 127 | }; |
| 128 | let (mut engine, _handle) = Engine::new(config, &Config::default()); |
| 129 | let context = engine.installed_next_turn_prompt_context(); |
| 130 | |
| 131 | // Turn 1: establishes the pin key; nothing to report. |
| 132 | assert_eq!(engine.refresh_pinned_header_for_turn(&context), None); |
| 133 | let pinned = engine.session.system_prompt.clone(); |
| 134 | let pinned_hash = engine.session.last_system_prompt_hash; |
| 135 | engine.session.pending_prefix_change_reason = None; |
| 136 | |
| 137 | // No change → no snapshot. |
| 138 | assert_eq!(engine.refresh_pinned_header_for_turn(&context), None); |
| 139 | |
| 140 | // The agent writes a file "mid-turn"; nothing moves until the next user turn. |
| 141 | fs::write(tmp.path().join("NEWFILE.md"), "brand new content").expect("write"); |
| 142 | assert_eq!(engine.session.system_prompt, pinned); |
| 143 | |
| 144 | // Next user turn: header byte-identical, exactly one snapshot with the delta. |
| 145 | let update = engine |
| 146 | .refresh_pinned_header_for_turn(&context) |
| 147 | .expect("workspace drift produces a context update"); |
| 148 | assert!(update.starts_with("<context_update>"), "{update}"); |
| 149 | assert!(update.contains("NEWFILE.md"), "{update}"); |
| 150 | assert_eq!(engine.session.system_prompt, pinned); |
| 151 | assert_eq!(engine.session.last_system_prompt_hash, pinned_hash); |
| 152 | assert_eq!(engine.session.pending_prefix_change_reason, None); |
| 153 | assert_eq!( |
| 154 | engine |
| 155 | .session |
| 156 | .prefix_stability |
| 157 | .as_ref() |
| 158 | .unwrap() |
| 159 | .context_update_count(), |
| 160 | 1 |
| 161 | ); |
| 162 | |
| 163 | // The same delta is not re-sent on the following turn. |
| 164 | assert_eq!(engine.refresh_pinned_header_for_turn(&context), None); |
| 165 | } |
| 166 | |
| 167 | #[test] |
| 168 | fn agents_md_edit_arrives_as_context_update_carrying_the_new_instructions() { |
| 169 | let _lock = lock_test_env(); |
| 170 | let tmp = tempdir().expect("tempdir"); |
| 171 | fs::write(tmp.path().join("AGENTS.md"), "# Rules\n\nAlways run fmt.\n").expect("write"); |
| 172 | let config = EngineConfig { |
| 173 | workspace: tmp.path().to_path_buf(), |
| 174 | ..Default::default() |
| 175 | }; |
| 176 | let (mut engine, _handle) = Engine::new(config, &Config::default()); |
| 177 | let context = engine.installed_next_turn_prompt_context(); |
| 178 | assert_eq!(engine.refresh_pinned_header_for_turn(&context), None); |
| 179 | let pinned = engine.session.system_prompt.clone(); |
| 180 | |
| 181 | fs::write( |
| 182 | tmp.path().join("AGENTS.md"), |
| 183 | "# Rules\n\nAlways run fmt.\nNever push to main.\n", |
| 184 | ) |
| 185 | .expect("write"); |
| 186 | let update = engine |
| 187 | .refresh_pinned_header_for_turn(&context) |
| 188 | .expect("AGENTS.md edit produces a context update"); |
| 189 | assert!(update.contains("+ Never push to main."), "{update}"); |
| 190 | assert_eq!(engine.session.system_prompt, pinned); |
| 191 | } |
| 192 | |
| 193 | #[test] |
| 194 | fn explicit_input_change_repins_instead_of_snapshotting() { |
| 195 | let _lock = lock_test_env(); |
| 196 | let tmp = tempdir().expect("tempdir"); |
| 197 | let config = EngineConfig { |
| 198 | workspace: tmp.path().to_path_buf(), |
| 199 | ..Default::default() |
| 200 | }; |
| 201 | let (mut engine, _handle) = Engine::new(config, &Config::default()); |
| 202 | let context = engine.installed_next_turn_prompt_context(); |
| 203 | assert_eq!(engine.refresh_pinned_header_for_turn(&context), None); |
| 204 | engine.session.pending_prefix_change_reason = None; |
| 205 | |
| 206 | let mut next = context.clone(); |
| 207 | next.goal_objective = Some("ship 0.9.8".to_string()); |
| 208 | assert_eq!(engine.refresh_pinned_header_for_turn(&next), None); |
| 209 | assert_eq!( |
| 210 | engine.session.pending_prefix_change_reason.as_deref(), |
| 211 | Some("goal") |
| 212 | ); |
| 213 | assert_eq!(engine.session.pinned_prompt_context.as_ref(), Some(&next)); |
| 214 | } |
| 215 | |
| 216 | #[tokio::test] |
| 217 | async fn submitted_turn_appends_context_update_before_the_user_message() { |
| 218 | let _lock = lock_test_env(); |
| 219 | let tmp = tempdir().expect("tempdir"); |
| 220 | let config = EngineConfig { |
| 221 | workspace: tmp.path().to_path_buf(), |
| 222 | project_context_pack_enabled: true, |
| 223 | ..Default::default() |
| 224 | }; |
| 225 | let (mut engine, _handle) = Engine::new(config, &Config::default()); |
| 226 | // Seed the pin key exactly as the first turn would. |
| 227 | let context = engine.installed_next_turn_prompt_context(); |
| 228 | assert_eq!(engine.refresh_pinned_header_for_turn(&context), None); |
| 229 | fs::write(tmp.path().join("NEWFILE.md"), "brand new content").expect("write"); |
| 230 | |
| 231 | // Drive the real submit path (no client → it stops before any request, |
| 232 | // but only after history is assembled) and inspect the order. |
| 233 | let update = engine.refresh_pinned_header_for_turn(&context).unwrap(); |
| 234 | engine.session.add_message(Message { |
| 235 | role: Role::User, |
| 236 | content: vec![ContentBlock::Text { |
| 237 | text: update, |
| 238 | cache_control: None, |
| 239 | }], |
| 240 | }); |
| 241 | engine |
| 242 | .session |
| 243 | .add_message(engine.user_text_message_with_turn_metadata("hello".into())); |
| 244 | let updates = context_update_messages(&engine); |
| 245 | assert_eq!(updates.len(), 1); |
| 246 | let last_two: Vec<&Message> = engine.session.messages.iter().rev().take(2).collect(); |
| 247 | assert!(matches!( |
| 248 | last_two[1].content.first(), |
| 249 | Some(ContentBlock::Text { text, .. }) if text.starts_with("<context_update>") |
| 250 | )); |
| 251 | assert!(matches!( |
| 252 | last_two[0].content.first(), |
| 253 | Some(ContentBlock::Text { text, .. }) if text == "hello" |
| 254 | )); |
| 255 | } |
| 256 | |
| 257 | #[test] |
| 258 | fn engine_prompt_keeps_reasoning_on_the_user_language_contract() { |
| 259 | let tmp = tempdir().expect("tempdir"); |
| 260 | let config = EngineConfig { |
| 261 | workspace: tmp.path().to_path_buf(), |
| 262 | locale_tag: "zh-Hans".to_string(), |
| 263 | ..Default::default() |
| 264 | }; |
| 265 | let (engine, _handle) = Engine::new(config, &Config::default()); |
| 266 | let prompt = match engine.session.system_prompt.as_ref() { |
| 267 | Some(SystemPrompt::Text(text)) => text.clone(), |
| 268 | Some(SystemPrompt::Blocks(blocks)) => blocks |
| 269 | .iter() |
| 270 | .map(|block| block.text.as_str()) |
| 271 | .collect::<Vec<_>>() |
| 272 | .join("\n\n"), |
| 273 | None => panic!("expected system prompt"), |
| 274 | }; |
| 275 | |
| 276 | assert!(prompt.contains("## Language")); |
| 277 | assert!(prompt.contains("latest\nuser message")); |
| 278 | assert!(prompt.contains("reasoning_content")); |
| 279 | assert!(prompt.contains("## 语言再次提醒")); |
| 280 | assert!(!prompt.contains("## Hidden Thinking Language")); |
| 281 | } |
| 282 | |
| 283 | fn sync_runtime_system_prompt_override(engine: &mut Engine, system_prompt: SystemPrompt) { |
| 284 | engine.session.compaction_summary_prompt = |
| 285 | extract_compaction_summary_prompt(Some(system_prompt.clone())); |
| 286 | engine.session.system_prompt = Some(system_prompt); |
| 287 | engine.session.system_prompt_override = true; |
| 288 | } |
| 289 | |
| 290 | #[test] |
| 291 | fn text_system_prompt_override_via_runtime_sync_survives_refresh() { |
| 292 | let tmp = tempdir().expect("tempdir"); |
| 293 | let config = EngineConfig { |
| 294 | workspace: tmp.path().to_path_buf(), |
| 295 | ..Default::default() |
| 296 | }; |
| 297 | let (mut engine, _handle) = Engine::new(config, &Config::default()); |
| 298 | let prompt = SystemPrompt::Text("TANGERINE-7".to_string()); |
| 299 | let expected = Some(prompt.clone()); |
| 300 | |
| 301 | sync_runtime_system_prompt_override(&mut engine, prompt); |
| 302 | engine.refresh_system_prompt(); |
| 303 | |
| 304 | assert_eq!(engine.session.system_prompt, expected); |
| 305 | } |
| 306 | |
| 307 | #[test] |
| 308 | fn blocks_system_prompt_override_via_runtime_sync_survives_mode_change_refresh() { |
| 309 | let tmp = tempdir().expect("tempdir"); |
| 310 | let config = EngineConfig { |
| 311 | workspace: tmp.path().to_path_buf(), |
| 312 | ..Default::default() |
| 313 | }; |
| 314 | let (mut engine, _handle) = Engine::new(config, &Config::default()); |
| 315 | let prompt = SystemPrompt::Blocks(vec![SystemBlock { |
| 316 | block_type: "text".to_string(), |
| 317 | text: "TANGERINE-7".to_string(), |
| 318 | cache_control: None, |
| 319 | }]); |
| 320 | let expected = Some(prompt.clone()); |
| 321 | |
| 322 | sync_runtime_system_prompt_override(&mut engine, prompt); |
| 323 | engine.refresh_system_prompt(); |
| 324 | |
| 325 | assert_eq!(engine.session.system_prompt, expected); |
| 326 | } |
| 327 | |
| 328 | #[test] |
| 329 | fn compaction_checkpoint_stays_out_of_stable_system_prompt() { |
| 330 | let tmp = tempdir().expect("tempdir"); |
| 331 | fs::create_dir_all(tmp.path().join("src")).expect("mkdir"); |
| 332 | fs::write(tmp.path().join("src/main.rs"), "fn main() {}").expect("write"); |
| 333 | |
| 334 | let config = EngineConfig { |
| 335 | workspace: tmp.path().to_path_buf(), |
| 336 | ..Default::default() |
| 337 | }; |
| 338 | let (mut engine, _handle) = Engine::new(config, &Config::default()); |
| 339 | engine |
| 340 | .session |
| 341 | .working_set |
| 342 | .observe_user_message("continue in src/main.rs", tmp.path()); |
| 343 | engine.refresh_system_prompt(); |
| 344 | let stable_before = engine.session.system_prompt.clone(); |
| 345 | engine.commit_compaction_checkpoint(Some(SystemPrompt::Blocks(vec![SystemBlock { |
| 346 | block_type: "text".to_string(), |
| 347 | text: format!("{COMPACTION_SUMMARY_MARKER}\nsummary"), |
| 348 | cache_control: None, |
| 349 | }]))); |
| 350 | |
| 351 | let prompt = match &engine.session.system_prompt { |
| 352 | Some(SystemPrompt::Text(text)) => text.clone(), |
| 353 | Some(SystemPrompt::Blocks(blocks)) => blocks |
| 354 | .iter() |
| 355 | .map(|block| block.text.as_str()) |
| 356 | .collect::<Vec<_>>() |
| 357 | .join("\n"), |
| 358 | None => panic!("expected system prompt"), |
| 359 | }; |
| 360 | |
| 361 | assert_eq!(engine.session.system_prompt, stable_before); |
| 362 | assert!(!prompt.contains(COMPACTION_SUMMARY_MARKER)); |
| 363 | assert!(!prompt.contains(WORKING_SET_SUMMARY_MARKER)); |
| 364 | assert!( |
| 365 | engine |
| 366 | .rendered_compaction_summary() |
| 367 | .expect("checkpoint") |
| 368 | .contains("summary") |
| 369 | ); |
| 370 | } |
| 371 | |
| 372 | /// Repeated compaction replaces the host-persistence copy while the stable |
| 373 | /// system prefix remains byte-for-byte unchanged. |
| 374 | #[test] |
| 375 | fn repeated_compaction_replaces_checkpoint_without_prefix_churn() { |
| 376 | let (mut engine, _handle) = Engine::new(EngineConfig::default(), &Config::default()); |
| 377 | engine.session.system_prompt = Some(SystemPrompt::Text("stable base prompt".to_string())); |
| 378 | |
| 379 | let flatten = |prompt: &Option<SystemPrompt>| match prompt { |
| 380 | Some(SystemPrompt::Text(text)) => text.clone(), |
| 381 | Some(SystemPrompt::Blocks(blocks)) => blocks |
| 382 | .iter() |
| 383 | .map(|block| block.text.as_str()) |
| 384 | .collect::<Vec<_>>() |
| 385 | .join("\n"), |
| 386 | None => String::new(), |
| 387 | }; |
| 388 | |
| 389 | let stable_before = engine.session.system_prompt.clone(); |
| 390 | for round in 0..3 { |
| 391 | engine.commit_compaction_checkpoint(Some(SystemPrompt::Text(format!( |
| 392 | "{COMPACTION_SUMMARY_MARKER}\nround-{round} summary body" |
| 393 | )))); |
| 394 | assert_eq!(engine.session.system_prompt, stable_before); |
| 395 | let prompt = flatten(&engine.session.compaction_summary_prompt); |
| 396 | assert_eq!( |
| 397 | prompt.matches(COMPACTION_SUMMARY_MARKER).count(), |
| 398 | 1, |
| 399 | "round {round}: exactly one checkpoint: {prompt}" |
| 400 | ); |
| 401 | assert!( |
| 402 | prompt.contains(&format!("round-{round} summary body")), |
| 403 | "{prompt}" |
| 404 | ); |
| 405 | } |
| 406 | } |
| 407 | |
| 408 | #[test] |
| 409 | fn caller_policy_defaults_to_direct() { |
| 410 | let tool = Tool { |
| 411 | tool_type: None, |
| 412 | name: "read_file".to_string(), |
| 413 | description: "Read".to_string(), |
| 414 | input_schema: json!({"type":"object"}), |
| 415 | allowed_callers: Some(vec!["direct".to_string()]), |
| 416 | defer_loading: Some(false), |
| 417 | input_examples: None, |
| 418 | strict: None, |
| 419 | cache_control: None, |
| 420 | }; |
| 421 | let direct = ToolCaller { |
| 422 | caller_type: "direct".to_string(), |
| 423 | tool_id: None, |
| 424 | }; |
| 425 | let code = ToolCaller { |
| 426 | caller_type: "code_execution_20250825".to_string(), |
| 427 | tool_id: Some("srvtoolu_1".to_string()), |
| 428 | }; |
| 429 | assert!(caller_allowed_for_tool(Some(&direct), Some(&tool))); |
| 430 | assert!(!caller_allowed_for_tool(Some(&code), Some(&tool))); |
| 431 | assert!(caller_allowed_for_tool(None, Some(&tool))); |
| 432 | } |
| 433 | |
| 434 | #[test] |
| 435 | fn tool_search_activates_discovered_deferred_tools() { |
| 436 | let mut catalog = vec![ |
| 437 | Tool { |
| 438 | tool_type: None, |
| 439 | name: "read_file".to_string(), |
| 440 | description: "Read files".to_string(), |
| 441 | input_schema: json!({"type":"object","properties":{"path":{"type":"string"}}}), |
| 442 | allowed_callers: Some(vec!["direct".to_string()]), |
| 443 | defer_loading: Some(true), |
| 444 | input_examples: None, |
| 445 | strict: None, |
| 446 | cache_control: None, |
| 447 | }, |
| 448 | Tool { |
| 449 | tool_type: None, |
| 450 | name: "grep_files".to_string(), |
| 451 | description: "Search files".to_string(), |
| 452 | input_schema: json!({"type":"object","properties":{"pattern":{"type":"string"}}}), |
| 453 | allowed_callers: Some(vec!["direct".to_string()]), |
| 454 | defer_loading: Some(true), |
| 455 | input_examples: None, |
| 456 | strict: None, |
| 457 | cache_control: None, |
| 458 | }, |
| 459 | ]; |
| 460 | let always_load = HashSet::new(); |
| 461 | ensure_advanced_tooling( |
| 462 | &mut catalog, |
| 463 | AppMode::Agent, |
| 464 | &always_load, |
| 465 | crate::core::engine::tool_catalog::ToolMode::Direct, |
| 466 | ); |
| 467 | let mut active = initial_active_tools(&catalog); |
| 468 | let result = execute_tool_search( |
| 469 | TOOL_SEARCH_NAME, |
| 470 | &json!({"query":"read file"}), |
| 471 | &catalog, |
| 472 | &mut active, |
| 473 | ) |
| 474 | .expect("search succeeds"); |
| 475 | assert!(result.success); |
| 476 | assert!(active.contains("read_file")); |
| 477 | } |
| 478 | |
| 479 | #[test] |
| 480 | fn tool_search_scenario() { |
| 481 | // Scenario consolidation of: tool_search_can_discover_request_user_input_modal_tool, tool_search_defaults_to_eight_results_for_regex_and_bm25, tool_search_respects_and_caps_max_results, tool_search_schema_exposes_max_results_default_and_cap |
| 482 | // from tool_search_can_discover_request_user_input_modal_tool |
| 483 | { |
| 484 | let always_load = HashSet::new(); |
| 485 | let mut catalog = build_model_tool_catalog( |
| 486 | vec![api_tool(REQUEST_USER_INPUT_NAME)], |
| 487 | Vec::new(), |
| 488 | AppMode::Agent, |
| 489 | &always_load, |
| 490 | ); |
| 491 | ensure_advanced_tooling( |
| 492 | &mut catalog, |
| 493 | AppMode::Agent, |
| 494 | &always_load, |
| 495 | crate::core::engine::tool_catalog::ToolMode::Direct, |
| 496 | ); |
| 497 | |
| 498 | let mut active = initial_active_tools(&catalog); |
| 499 | assert!(!active.contains(REQUEST_USER_INPUT_NAME)); |
| 500 | |
| 501 | let result = execute_tool_search( |
| 502 | TOOL_SEARCH_NAME, |
| 503 | &json!({"query":"ask user question"}), |
| 504 | &catalog, |
| 505 | &mut active, |
| 506 | ) |
| 507 | .expect("search succeeds"); |
| 508 | |
| 509 | assert!(result.success); |
| 510 | assert!(active.contains(REQUEST_USER_INPUT_NAME)); |
| 511 | } |
| 512 | // from tool_search_defaults_to_eight_results_for_regex_and_bm25 |
| 513 | { |
| 514 | let catalog = tool_search_catalog_with_matches(25); |
| 515 | |
| 516 | for match_kind in ["regex", "bm25"] { |
| 517 | let mut active = initial_active_tools(&catalog); |
| 518 | let result = execute_tool_search( |
| 519 | TOOL_SEARCH_NAME, |
| 520 | &json!({"query":"matching","match":match_kind}), |
| 521 | &catalog, |
| 522 | &mut active, |
| 523 | ) |
| 524 | .expect("search succeeds"); |
| 525 | |
| 526 | assert_eq!(tool_search_reference_count(&result), 8); |
| 527 | } |
| 528 | } |
| 529 | // from tool_search_respects_and_caps_max_results |
| 530 | { |
| 531 | let catalog = tool_search_catalog_with_matches(120); |
| 532 | |
| 533 | let mut active = initial_active_tools(&catalog); |
| 534 | let limited = execute_tool_search( |
| 535 | TOOL_SEARCH_NAME, |
| 536 | &json!({"query":"matching","max_results":7}), |
| 537 | &catalog, |
| 538 | &mut active, |
| 539 | ) |
| 540 | .expect("search succeeds"); |
| 541 | assert_eq!(tool_search_reference_count(&limited), 7); |
| 542 | |
| 543 | let mut active = initial_active_tools(&catalog); |
| 544 | let capped = execute_tool_search( |
| 545 | TOOL_SEARCH_NAME, |
| 546 | &json!({"query":"matching","match":"regex","max_results":999}), |
| 547 | &catalog, |
| 548 | &mut active, |
| 549 | ) |
| 550 | .expect("search succeeds"); |
| 551 | assert_eq!(tool_search_reference_count(&capped), 8); |
| 552 | } |
| 553 | // from tool_search_schema_exposes_max_results_default_and_cap |
| 554 | { |
| 555 | let mut catalog = Vec::new(); |
| 556 | let always_load = HashSet::new(); |
| 557 | ensure_advanced_tooling( |
| 558 | &mut catalog, |
| 559 | AppMode::Agent, |
| 560 | &always_load, |
| 561 | crate::core::engine::tool_catalog::ToolMode::Direct, |
| 562 | ); |
| 563 | |
| 564 | let tool = catalog |
| 565 | .iter() |
| 566 | .find(|tool| tool.name == TOOL_SEARCH_NAME) |
| 567 | .expect("tool search definition exists"); |
| 568 | let schema = &tool.input_schema["properties"]["max_results"]; |
| 569 | |
| 570 | assert_eq!(schema["default"], 8); |
| 571 | assert_eq!(schema["maximum"], 8); |
| 572 | assert_eq!(schema["minimum"], 1); |
| 573 | assert_eq!(tool.input_schema["properties"]["match"]["default"], "bm25"); |
| 574 | } |
| 575 | } |
| 576 | |
| 577 | fn tool_search_catalog_with_matches(count: usize) -> Vec<Tool> { |
| 578 | let mut catalog = (0..count) |
| 579 | .map(|idx| Tool { |
| 580 | tool_type: None, |
| 581 | name: format!("matching_tool_{idx:03}"), |
| 582 | description: "Matching deferred test tool".to_string(), |
| 583 | input_schema: json!({"type":"object","properties":{"query":{"type":"string"}}}), |
| 584 | allowed_callers: Some(vec!["direct".to_string()]), |
| 585 | defer_loading: Some(true), |
| 586 | input_examples: None, |
| 587 | strict: None, |
| 588 | cache_control: None, |
| 589 | }) |
| 590 | .collect::<Vec<_>>(); |
| 591 | let always_load = HashSet::new(); |
| 592 | ensure_advanced_tooling( |
| 593 | &mut catalog, |
| 594 | AppMode::Agent, |
| 595 | &always_load, |
| 596 | crate::core::engine::tool_catalog::ToolMode::Direct, |
| 597 | ); |
| 598 | catalog |
| 599 | } |
| 600 | |
| 601 | fn tool_search_reference_count(result: &ToolResult) -> usize { |
| 602 | result |
| 603 | .metadata |
| 604 | .as_ref() |
| 605 | .and_then(|metadata| metadata.get("tool_references")) |
| 606 | .and_then(|references| references.as_array()) |
| 607 | .map_or(0, Vec::len) |
| 608 | } |
| 609 | |
| 610 | #[tokio::test] |
| 611 | async fn execute_tools_dispatches_through_common_executor() { |
| 612 | use crate::tools::file_tool::ReadTool; |
| 613 | use crate::tools::registry::ToolRegistryBuilder; |
| 614 | use crate::tools::spec::ToolContext; |
| 615 | |
| 616 | let tmp = tempdir().expect("tempdir"); |
| 617 | std::fs::write(tmp.path().join("note.txt"), "alpha\n").expect("write note"); |
| 618 | let context = ToolContext::new(tmp.path()); |
| 619 | let registry = ToolRegistryBuilder::new() |
| 620 | .with_tool(Arc::new(ReadTool)) |
| 621 | .build(context.clone()); |
| 622 | let path = tmp |
| 623 | .path() |
| 624 | .join("note.txt") |
| 625 | .to_string_lossy() |
| 626 | .replace('\\', "\\\\"); |
| 627 | let code = format!( |
| 628 | "const r = await tools.call('read', {{ path: '{path}' }}); return JSON.stringify(r).includes('alpha');" |
| 629 | ); |
| 630 | let (tx_event, _rx_event) = mpsc::channel(8); |
| 631 | let result = Engine::execute_tool_with_lock( |
| 632 | Arc::new(RwLock::new(())), |
| 633 | false, |
| 634 | false, |
| 635 | tx_event, |
| 636 | None, |
| 637 | EXECUTE_TOOLS_TOOL_NAME.to_string(), |
| 638 | None, |
| 639 | json!({"code": code}), |
| 640 | tmp.path().to_path_buf(), |
| 641 | Some(®istry), |
| 642 | None, |
| 643 | Some(context), |
| 644 | ) |
| 645 | .await |
| 646 | .expect("execute_tools should dispatch"); |
| 647 | assert!(result.content.contains("\"nested_calls\":1")); |
| 648 | assert!(result.content.contains("true")); |
| 649 | } |
| 650 | |
| 651 | #[tokio::test] |
| 652 | async fn dispatch_reports_typed_operation_activity_without_names_or_arguments() { |
| 653 | use crate::tools::file_tool::ReadTool; |
| 654 | use crate::tools::registry::ToolRegistryBuilder; |
| 655 | use crate::tools::spec::ToolContext; |
| 656 | use codewhale_protocol::engine_owner::{OwnerActivityKind, OwnerOperationOutcome}; |
| 657 | |
| 658 | let tmp = tempdir().expect("tempdir"); |
| 659 | std::fs::write(tmp.path().join("private-note.txt"), "alpha\n").expect("write note"); |
| 660 | let context = ToolContext::new(tmp.path()); |
| 661 | let registry = ToolRegistryBuilder::new() |
| 662 | .with_tool(Arc::new(ReadTool)) |
| 663 | .build(context.clone()); |
| 664 | |
| 665 | let run = |name: &'static str, span: Option<&'static str>, input: serde_json::Value| { |
| 666 | let registry = ®istry; |
| 667 | let context = context.clone(); |
| 668 | let workspace = tmp.path().to_path_buf(); |
| 669 | async move { |
| 670 | let (tx_event, mut rx_event) = mpsc::channel(16); |
| 671 | let _ = Engine::execute_tool_with_lock( |
| 672 | Arc::new(RwLock::new(())), |
| 673 | false, |
| 674 | false, |
| 675 | tx_event, |
| 676 | None, |
| 677 | name.to_string(), |
| 678 | span.map(str::to_string), |
| 679 | input, |
| 680 | workspace, |
| 681 | Some(registry), |
| 682 | None, |
| 683 | Some(context), |
| 684 | ) |
| 685 | .await; |
| 686 | let mut events = Vec::new(); |
| 687 | while let Ok(event) = rx_event.try_recv() { |
| 688 | if matches!( |
| 689 | event, |
| 690 | Event::OperationActivityStarted { .. } |
| 691 | | Event::OperationActivityCompleted { .. } |
| 692 | ) { |
| 693 | events.push(event); |
| 694 | } |
| 695 | } |
| 696 | events |
| 697 | } |
| 698 | }; |
| 699 | |
| 700 | let events = run("read", Some("call-1"), json!({"path": "private-note.txt"})).await; |
| 701 | assert!( |
| 702 | matches!( |
| 703 | events.as_slice(), |
| 704 | [ |
| 705 | Event::OperationActivityStarted { span_id: started, activity_kind: OwnerActivityKind::Reading }, |
| 706 | Event::OperationActivityCompleted { |
| 707 | span_id: completed, |
| 708 | activity_kind: OwnerActivityKind::Reading, |
| 709 | outcome: OwnerOperationOutcome::Succeeded, |
| 710 | }, |
| 711 | ] if started == completed && started.starts_with("call-1#") |
| 712 | ), |
| 713 | "unexpected activity: {events:?}" |
| 714 | ); |
| 715 | // A repeated model call id (gateways that elide ids fall back to |
| 716 | // `call_{block_index}`) still gets a fresh span, so a consumer that |
| 717 | // deduplicates completed spans sees the second call. |
| 718 | let again = run("read", Some("call-1"), json!({"path": "private-note.txt"})).await; |
| 719 | let span_of = |events: &[Event]| match events.first() { |
| 720 | Some(Event::OperationActivityStarted { span_id, .. }) => span_id.clone(), |
| 721 | other => panic!("unexpected activity: {other:?}"), |
| 722 | }; |
| 723 | assert_ne!(span_of(&events), span_of(&again)); |
| 724 | let wire = format!("{events:?}"); |
| 725 | assert!(!wire.contains("private-note"), "arguments leaked: {wire}"); |
| 726 | |
| 727 | // A failed read still reports its kind, with a typed outcome only. |
| 728 | let events = run("read", Some("call-2"), json!({"path": "missing.txt"})).await; |
| 729 | assert!( |
| 730 | matches!( |
| 731 | events.last(), |
| 732 | Some(Event::OperationActivityCompleted { |
| 733 | outcome: OwnerOperationOutcome::Failed, |
| 734 | .. |
| 735 | }) |
| 736 | ), |
| 737 | "unexpected activity: {events:?}" |
| 738 | ); |
| 739 | |
| 740 | // No span id (internal/unattributed dispatch), an unregistered name, and |
| 741 | // the code-mode wrapper itself report nothing. |
| 742 | assert!( |
| 743 | run("read", None, json!({"path": "private-note.txt"})) |
| 744 | .await |
| 745 | .is_empty() |
| 746 | ); |
| 747 | assert!( |
| 748 | run("not_a_tool", Some("call-3"), json!({})) |
| 749 | .await |
| 750 | .is_empty() |
| 751 | ); |
| 752 | assert!( |
| 753 | run( |
| 754 | EXECUTE_TOOLS_TOOL_NAME, |
| 755 | Some("call-4"), |
| 756 | json!({"code": "return 1;"}) |
| 757 | ) |
| 758 | .await |
| 759 | .is_empty() |
| 760 | ); |
| 761 | |
| 762 | // A call refused by the cancel gate never ran, so it reports nothing. |
| 763 | let (tx_event, mut rx_event) = mpsc::channel(16); |
| 764 | let cancelled = CancellationToken::new(); |
| 765 | cancelled.cancel(); |
| 766 | let refused = Engine::execute_tool_with_lock( |
| 767 | Arc::new(RwLock::new(())), |
| 768 | false, |
| 769 | false, |
| 770 | tx_event, |
| 771 | Some(cancelled), |
| 772 | "read".to_string(), |
| 773 | Some("call-5".to_string()), |
| 774 | json!({"path": "private-note.txt"}), |
| 775 | tmp.path().to_path_buf(), |
| 776 | Some(®istry), |
| 777 | None, |
| 778 | Some(context.clone()), |
| 779 | ) |
| 780 | .await; |
| 781 | assert!(refused.is_err()); |
| 782 | while let Ok(event) = rx_event.try_recv() { |
| 783 | assert!( |
| 784 | !matches!( |
| 785 | event, |
| 786 | Event::OperationActivityStarted { .. } | Event::OperationActivityCompleted { .. } |
| 787 | ), |
| 788 | "a refused call reported activity: {event:?}" |
| 789 | ); |
| 790 | } |
| 791 | } |
| 792 | |
| 793 | #[tokio::test] |
| 794 | async fn dropped_operation_span_completes_as_cancelled() { |
| 795 | use codewhale_protocol::engine_owner::{OwnerActivityKind, OwnerOperationOutcome}; |
| 796 | |
| 797 | // The turn loop drops an in-flight tool future on cancel; the span it |
| 798 | // opened must still close, or every host leaks an active operation. |
| 799 | let (tx_event, mut rx_event) = mpsc::channel(16); |
| 800 | let span = super::tool_execution::OperationSpanGuard::start( |
| 801 | tx_event, |
| 802 | "call-x", |
| 803 | OwnerActivityKind::Editing, |
| 804 | None, |
| 805 | ) |
| 806 | .await; |
| 807 | drop(span); |
| 808 | let started = match rx_event.try_recv() { |
| 809 | Ok(Event::OperationActivityStarted { span_id, .. }) => span_id, |
| 810 | other => panic!("unexpected event: {other:?}"), |
| 811 | }; |
| 812 | match rx_event.try_recv() { |
| 813 | Ok(Event::OperationActivityCompleted { |
| 814 | span_id, |
| 815 | activity_kind: OwnerActivityKind::Editing, |
| 816 | outcome: OwnerOperationOutcome::Cancelled, |
| 817 | }) => assert_eq!(span_id, started), |
| 818 | other => panic!("unexpected event: {other:?}"), |
| 819 | } |
| 820 | assert!(rx_event.try_recv().is_err(), "exactly one Completed"); |
| 821 | } |
| 822 | |
| 823 | #[tokio::test] |
| 824 | async fn code_execution_does_not_inherit_parent_secret_env() { |
| 825 | use crate::dependencies::ExternalTool as _; |
| 826 | if !crate::dependencies::Python::available() { |
| 827 | // `dependencies::tests::runtime_commands_do_not_inherit_parent_secret_env` |
| 828 | // still covers the scrubbed Python constructor without Python. |
| 829 | return; |
| 830 | } |
| 831 | let _env_lock = lock_test_env(); |
| 832 | let _secret = EnvVarGuard::set("CODEWHALE_TEST_FAKE_API_KEY", "sk-test-sentinel"); |
| 833 | let tmp = tempdir().expect("tempdir"); |
| 834 | let result = execute_code_execution_tool( |
| 835 | &json!({"code":"import os; print(os.environ.get('CODEWHALE_TEST_FAKE_API_KEY', 'absent'))"}), |
| 836 | tmp.path(), |
| 837 | &crate::tools::spec::ToolContext::new(tmp.path()), |
| 838 | ) |
| 839 | .await |
| 840 | .expect("code execution should run"); |
| 841 | let stdout = result.metadata.as_ref().expect("payload")["stdout"] |
| 842 | .as_str() |
| 843 | .unwrap_or_default() |
| 844 | .to_string(); |
| 845 | assert_eq!(stdout.trim(), "absent", "{}", result.content); |
| 846 | assert!(!result.content.contains("sk-test-sentinel")); |
| 847 | } |
| 848 | |
| 849 | #[tokio::test] |
| 850 | async fn code_execution_scenario() { |
| 851 | // Scenario consolidation of: code_execution_runs_python_and_returns_result_payload, code_execution_runs_through_common_executor_after_approval_gate |
| 852 | // from code_execution_runs_python_and_returns_result_payload |
| 853 | { |
| 854 | let tmp = tempdir().expect("tempdir"); |
| 855 | let result = execute_code_execution_tool( |
| 856 | &json!({"code":"print('hello from code exec')"}), |
| 857 | tmp.path(), |
| 858 | &crate::tools::spec::ToolContext::new(tmp.path()), |
| 859 | ) |
| 860 | .await |
| 861 | .expect("code execution should run"); |
| 862 | assert!(result.content.contains("hello from code exec")); |
| 863 | assert!(result.content.contains("return_code")); |
| 864 | } |
| 865 | // from code_execution_runs_through_common_executor_after_approval_gate |
| 866 | { |
| 867 | let tmp = tempdir().expect("tempdir"); |
| 868 | let (tx_event, _rx_event) = mpsc::channel(8); |
| 869 | let result = Engine::execute_tool_with_lock( |
| 870 | Arc::new(RwLock::new(())), |
| 871 | false, |
| 872 | false, |
| 873 | tx_event, |
| 874 | None, |
| 875 | CODE_EXECUTION_TOOL_NAME.to_string(), |
| 876 | None, |
| 877 | json!({"code":"print('common executor code exec')"}), |
| 878 | tmp.path().to_path_buf(), |
| 879 | None, |
| 880 | None, |
| 881 | Some(crate::tools::spec::ToolContext::new(tmp.path())), |
| 882 | ) |
| 883 | .await |
| 884 | .expect("code_execution should run through common executor"); |
| 885 | |
| 886 | assert!(result.result.content.contains("common executor code exec")); |
| 887 | assert!(result.result.content.contains("return_code")); |
| 888 | } |
| 889 | } |
| 890 | |
| 891 | #[tokio::test] |
| 892 | async fn interpreter_execution_policy_preserves_readonly_and_exact_call_elevation() { |
| 893 | use crate::sandbox::SandboxPolicy; |
| 894 | use crate::tools::spec::ToolContext; |
| 895 | for tool in [CODE_EXECUTION_TOOL_NAME, JS_EXECUTION_TOOL_NAME] { |
| 896 | let workspace = tempdir().expect("workspace"); |
| 897 | let outside = tempdir().expect("outside workspace"); |
| 898 | let context = ToolContext::new(workspace.path()) |
| 899 | .with_elevated_sandbox_policy(SandboxPolicy::ReadOnly); |
| 900 | for (name, policy, allowed) in [ |
| 901 | ("ordinary.txt", SandboxPolicy::ReadOnly, false), |
| 902 | ("approved.txt", SandboxPolicy::DangerFullAccess, true), |
| 903 | ("ordinary-again.txt", SandboxPolicy::ReadOnly, false), |
| 904 | ] { |
| 905 | let path = outside.path().join(name); |
| 906 | let literal = json!(path.to_string_lossy()).to_string(); |
| 907 | let code = if tool == CODE_EXECUTION_TOOL_NAME { |
| 908 | format!("open({literal}, 'w').write('proof')") |
| 909 | } else { |
| 910 | format!("require('fs').writeFileSync({literal}, 'proof')") |
| 911 | }; |
| 912 | let result = Engine::execute_tool_with_lock( |
| 913 | Arc::new(RwLock::new(())), |
| 914 | false, |
| 915 | false, |
| 916 | mpsc::channel(8).0, |
| 917 | None, |
| 918 | tool.to_string(), |
| 919 | None, |
| 920 | json!({"code": code}), |
| 921 | workspace.path().to_path_buf(), |
| 922 | None, |
| 923 | None, |
| 924 | Some(context.clone().with_elevated_sandbox_policy(policy)), |
| 925 | ) |
| 926 | .await; |
| 927 | if allowed { |
| 928 | let result = result.expect("explicitly elevated execution"); |
| 929 | assert!(result.result.success, "{tool}: {:?}", result.result); |
| 930 | } else if let Ok(result) = result { |
| 931 | assert!(!result.result.success, "{tool} ran a read-only write"); |
| 932 | } |
| 933 | assert_eq!(path.exists(), allowed, "{tool}: {name}"); |
| 934 | } |
| 935 | } |
| 936 | } |
| 937 | |
| 938 | #[tokio::test] |
| 939 | async fn interpreter_execution_policy_refuses_external_backend_local_fallback() { |
| 940 | use crate::tools::spec::ToolContext; |
| 941 | let workspace = tempdir().expect("workspace"); |
| 942 | let backend = crate::sandbox::backend::create_backend(&Config { |
| 943 | sandbox_backend: Some("unsupported-regression-backend".to_string()), |
| 944 | ..Config::default() |
| 945 | }) |
| 946 | .expect("backend policy") |
| 947 | .expect("external boundary"); |
| 948 | let context = ToolContext::new(workspace.path()).with_sandbox_backend(Arc::from(backend)); |
| 949 | for tool in [CODE_EXECUTION_TOOL_NAME, JS_EXECUTION_TOOL_NAME] { |
| 950 | let code = if tool == CODE_EXECUTION_TOOL_NAME { |
| 951 | "open('external-fallback.txt', 'w').write('bad')" |
| 952 | } else { |
| 953 | "require('fs').writeFileSync('external-fallback.txt', 'bad')" |
| 954 | }; |
| 955 | let error = Engine::execute_tool_with_lock( |
| 956 | Arc::new(RwLock::new(())), |
| 957 | false, |
| 958 | false, |
| 959 | mpsc::channel(8).0, |
| 960 | None, |
| 961 | tool.to_string(), |
| 962 | None, |
| 963 | json!({"code": code}), |
| 964 | workspace.path().to_path_buf(), |
| 965 | None, |
| 966 | None, |
| 967 | Some(context.clone()), |
| 968 | ) |
| 969 | .await |
| 970 | .expect_err("external backend cannot run a local interpreter"); |
| 971 | assert!(error.to_string().contains("external sandbox"), "{error}"); |
| 972 | } |
| 973 | assert!(!workspace.path().join("external-fallback.txt").exists()); |
| 974 | } |
| 975 | |
| 976 | #[test] |
| 977 | fn plan_mode_catalog_skips_code_execution_tool_but_agent_keeps_it() { |
| 978 | let mut plan_catalog = vec![api_tool("read_file")]; |
| 979 | let always_load = HashSet::new(); |
| 980 | ensure_advanced_tooling( |
| 981 | &mut plan_catalog, |
| 982 | AppMode::Plan, |
| 983 | &always_load, |
| 984 | crate::core::engine::tool_catalog::ToolMode::Direct, |
| 985 | ); |
| 986 | assert!( |
| 987 | !plan_catalog |
| 988 | .iter() |
| 989 | .any(|tool| tool.name == CODE_EXECUTION_TOOL_NAME), |
| 990 | "Plan mode must not expose code_execution" |
| 991 | ); |
| 992 | |
| 993 | let mut agent_catalog = vec![api_tool("read_file")]; |
| 994 | ensure_advanced_tooling( |
| 995 | &mut agent_catalog, |
| 996 | AppMode::Agent, |
| 997 | &always_load, |
| 998 | crate::core::engine::tool_catalog::ToolMode::Direct, |
| 999 | ); |
| 1000 | assert!( |
| 1001 | agent_catalog |
| 1002 | .iter() |
| 1003 | .any(|tool| tool.name == CODE_EXECUTION_TOOL_NAME), |
| 1004 | "Agent mode should still expose code_execution" |
| 1005 | ); |
| 1006 | } |
| 1007 | |
| 1008 | #[test] |
| 1009 | fn missing_tool_error_message_offers_suggestions() { |
| 1010 | let catalog = vec![ |
| 1011 | Tool { |
| 1012 | tool_type: None, |
| 1013 | name: "read_file".to_string(), |
| 1014 | description: "Read file contents".to_string(), |
| 1015 | input_schema: json!({"type":"object","properties":{"path":{"type":"string"}}}), |
| 1016 | allowed_callers: Some(vec!["direct".to_string()]), |
| 1017 | defer_loading: Some(false), |
| 1018 | input_examples: None, |
| 1019 | strict: None, |
| 1020 | cache_control: None, |
| 1021 | }, |
| 1022 | Tool { |
| 1023 | tool_type: None, |
| 1024 | name: "grep_files".to_string(), |
| 1025 | description: "Search file contents".to_string(), |
| 1026 | input_schema: json!({"type":"object","properties":{"pattern":{"type":"string"}}}), |
| 1027 | allowed_callers: Some(vec!["direct".to_string()]), |
| 1028 | defer_loading: Some(false), |
| 1029 | input_examples: None, |
| 1030 | strict: None, |
| 1031 | cache_control: None, |
| 1032 | }, |
| 1033 | ]; |
| 1034 | |
| 1035 | let message = missing_tool_error_message("reed_file", &catalog); |
| 1036 | assert!(message.contains("Did you mean:")); |
| 1037 | assert!(message.contains("read_file")); |
| 1038 | assert!(message.contains(TOOL_SEARCH_NAME)); |
| 1039 | } |
| 1040 | |
| 1041 | #[test] |
| 1042 | fn missing_tool_scenario() { |
| 1043 | // Scenario consolidation of: missing_tool_error_message_includes_discovery_guidance_when_no_match, missing_tool_error_message_redirects_checklist_item_miscalls, missing_tool_error_message_names_exec_shell_rename |
| 1044 | // from missing_tool_error_message_includes_discovery_guidance_when_no_match |
| 1045 | { |
| 1046 | let catalog = vec![Tool { |
| 1047 | tool_type: None, |
| 1048 | name: "read_file".to_string(), |
| 1049 | description: "Read file contents".to_string(), |
| 1050 | input_schema: json!({"type":"object","properties":{"path":{"type":"string"}}}), |
| 1051 | allowed_callers: Some(vec!["direct".to_string()]), |
| 1052 | defer_loading: Some(false), |
| 1053 | input_examples: None, |
| 1054 | strict: None, |
| 1055 | cache_control: None, |
| 1056 | }]; |
| 1057 | |
| 1058 | let message = missing_tool_error_message("totally_unknown_tool", &catalog); |
| 1059 | assert!(message.contains("not available in the current tool catalog")); |
| 1060 | assert!(message.contains(TOOL_SEARCH_NAME)); |
| 1061 | } |
| 1062 | // from missing_tool_error_message_redirects_checklist_item_miscalls |
| 1063 | { |
| 1064 | let catalog = vec![api_tool("note"), api_tool("tts")]; |
| 1065 | |
| 1066 | for tool_name in ["item", "items", "todo", "checklist_item"] { |
| 1067 | let message = missing_tool_error_message(tool_name, &catalog); |
| 1068 | assert!(message.contains("todo_write"), "{tool_name}: {message}"); |
| 1069 | assert!( |
| 1070 | !message.contains("Did you mean"), |
| 1071 | "fuzzy suggestions are misleading for checklist mis-calls: {message}" |
| 1072 | ); |
| 1073 | } |
| 1074 | } |
| 1075 | // from missing_tool_error_message_names_exec_shell_rename |
| 1076 | { |
| 1077 | // #5123-class: retired exec_shell must point at lowercase foreground bash, |
| 1078 | // not misdiagnosed as an allow_shell permission problem. |
| 1079 | let catalog = vec![api_tool("read_file")]; |
| 1080 | |
| 1081 | let message = missing_tool_error_message("exec_shell", &catalog); |
| 1082 | assert!(message.contains("replaced by `bash`"), "{message}"); |
| 1083 | assert!(message.contains("`command`"), "{message}"); |
| 1084 | |
| 1085 | for tool_name in [ |
| 1086 | "exec_shell_wait", |
| 1087 | "exec_shell_interact", |
| 1088 | "exec_shell_cancel", |
| 1089 | ] { |
| 1090 | let message = missing_tool_error_message(tool_name, &catalog); |
| 1091 | assert!(message.contains("not available in the current tool catalog")); |
| 1092 | assert!( |
| 1093 | message.contains("foreground-only"), |
| 1094 | "{tool_name}: {message}" |
| 1095 | ); |
| 1096 | assert!(message.contains(TOOL_SEARCH_NAME), "{tool_name}: {message}"); |
| 1097 | } |
| 1098 | } |
| 1099 | } |
| 1100 | |
| 1101 | #[test] |
| 1102 | fn missing_shell_scenario() { |
| 1103 | // Scenario consolidation of: missing_shell_tool_error_message_names_allow_shell_gate, missing_shell_tool_error_message_keeps_allow_shell_hint_with_suggestions |
| 1104 | // from missing_shell_tool_error_message_names_allow_shell_gate |
| 1105 | { |
| 1106 | let catalog = vec![api_tool("read_file")]; |
| 1107 | |
| 1108 | for tool_name in ["task_shell_start", "task_shell_wait"] { |
| 1109 | let message = missing_tool_error_message(tool_name, &catalog); |
| 1110 | assert!(message.contains("not available in the current tool catalog")); |
| 1111 | assert!( |
| 1112 | message.contains("allow_shell = false"), |
| 1113 | "{tool_name}: {message}" |
| 1114 | ); |
| 1115 | assert!(message.contains("allow_shell"), "{tool_name}: {message}"); |
| 1116 | assert!( |
| 1117 | message.contains("/config allow_shell true"), |
| 1118 | "{tool_name}: {message}" |
| 1119 | ); |
| 1120 | assert!(message.contains("--save"), "{tool_name}: {message}"); |
| 1121 | assert!(message.contains("Work mode"), "{tool_name}: {message}"); |
| 1122 | assert!( |
| 1123 | message.contains("approval gating"), |
| 1124 | "{tool_name}: {message}" |
| 1125 | ); |
| 1126 | assert!(!message.contains("YOLO"), "{tool_name}: {message}"); |
| 1127 | assert!(!message.contains("auto-approve"), "{tool_name}: {message}"); |
| 1128 | assert!(message.contains(TOOL_SEARCH_NAME), "{tool_name}: {message}"); |
| 1129 | } |
| 1130 | } |
| 1131 | // from missing_shell_tool_error_message_keeps_allow_shell_hint_with_suggestions |
| 1132 | { |
| 1133 | let catalog = vec![api_tool("task_shell_starter")]; |
| 1134 | |
| 1135 | let message = missing_tool_error_message("task_shell_start", &catalog); |
| 1136 | |
| 1137 | assert!(message.contains("Did you mean:")); |
| 1138 | assert!(message.contains("task_shell_starter")); |
| 1139 | assert!(message.contains("allow_shell = false")); |
| 1140 | assert!(message.contains("allow_shell")); |
| 1141 | assert!(message.contains("/config allow_shell true")); |
| 1142 | assert!(message.contains("--save")); |
| 1143 | assert!(message.contains("Work mode")); |
| 1144 | assert!(!message.contains("YOLO")); |
| 1145 | assert!(!message.contains("auto-approve")); |
| 1146 | assert!(message.contains(TOOL_SEARCH_NAME)); |
| 1147 | } |
| 1148 | } |
| 1149 | |
| 1150 | #[test] |
| 1151 | fn filter_tool_scenario() { |
| 1152 | // Scenario consolidation of: filter_tool_call_delta_strips_bracket_marker, filter_tool_call_delta_strips_deepseek_xml_marker, filter_tool_call_delta_strips_deepseek_native_tool_tokens, filter_tool_call_delta_strips_deepseek_native_token_split_across_chunks, filter_tool_call_delta_strips_generic_tool_call_marker, filter_tool_call_delta_strips_invoke_marker, filter_tool_call_delta_strips_function_calls_marker, filter_tool_call_delta_strips_siliconflow_v4_dsml_content_fixture |
| 1153 | // from filter_tool_call_delta_strips_bracket_marker |
| 1154 | { |
| 1155 | let mut in_block = false; |
| 1156 | let visible = filter_tool_call_delta( |
| 1157 | "intro [TOOL_CALL]\n{\"tool\":\"x\"}\n[/TOOL_CALL] outro", |
| 1158 | &mut in_block, |
| 1159 | ); |
| 1160 | assert!(!in_block); |
| 1161 | assert!(!visible.contains("[TOOL_CALL]")); |
| 1162 | assert!(!visible.contains("[/TOOL_CALL]")); |
| 1163 | assert!(!visible.contains("\"tool\":\"x\"")); |
| 1164 | assert!(visible.contains("intro")); |
| 1165 | assert!(visible.contains("outro")); |
| 1166 | } |
| 1167 | // from filter_tool_call_delta_strips_deepseek_xml_marker |
| 1168 | { |
| 1169 | let mut in_block = false; |
| 1170 | let visible = filter_tool_call_delta( |
| 1171 | "before <codewhale:tool_call name=\"x\">payload</codewhale:tool_call> after", |
| 1172 | &mut in_block, |
| 1173 | ); |
| 1174 | assert!(!in_block); |
| 1175 | for marker in TOOL_CALL_START_MARKERS { |
| 1176 | assert!( |
| 1177 | !visible.contains(marker), |
| 1178 | "visible text leaked start marker `{marker}`: {visible:?}" |
| 1179 | ); |
| 1180 | } |
| 1181 | assert!(visible.contains("before")); |
| 1182 | assert!(visible.contains("after")); |
| 1183 | } |
| 1184 | // from filter_tool_call_delta_strips_deepseek_native_tool_tokens |
| 1185 | { |
| 1186 | // #3880: DeepSeek's chat template separates words with `▁` (U+2581), so |
| 1187 | // `<|tool▁calls▁begin|>` matched no DSML entry and reached the user as |
| 1188 | // visible text that interrupted the task. |
| 1189 | for (start, end) in [ |
| 1190 | ("<|tool▁calls▁begin|>", "<|tool▁calls▁end|>"), |
| 1191 | ("<|tool▁call▁begin|>", "<|tool▁call▁end|>"), |
| 1192 | ("<|tool▁calls▁begin|>", "<|tool▁calls▁end|>"), |
| 1193 | ("<|tool_calls_begin|>", "<|tool_calls_end|>"), |
| 1194 | ("<|tool_call_begin|>", "<|tool_call_end|>"), |
| 1195 | ("<|tool▁outputs▁begin|>", "<|tool▁outputs▁end|>"), |
| 1196 | ] { |
| 1197 | let mut in_block = false; |
| 1198 | let visible = filter_tool_call_delta( |
| 1199 | &format!("before {start}function<|tool▁sep|>read_file\n{{}}{end} after"), |
| 1200 | &mut in_block, |
| 1201 | ); |
| 1202 | assert!(!in_block, "state stuck inside block for {start}"); |
| 1203 | assert!( |
| 1204 | !visible.contains("tool▁") && !visible.contains("tool_calls"), |
| 1205 | "leaked {start} into visible text: {visible:?}" |
| 1206 | ); |
| 1207 | assert!(visible.contains("before"), "{visible:?}"); |
| 1208 | assert!(visible.contains("after"), "{visible:?}"); |
| 1209 | } |
| 1210 | } |
| 1211 | // DeepSeek's doubled-delimiter DSML form, emitted when the request offers |
| 1212 | // no tools: one-shot `codewhale exec` printed it verbatim as the answer. |
| 1213 | { |
| 1214 | let text = "<||DSML|| calls>\n<||DSML|| invoke name=\"read_file\">\n<||DSML|| parameter name=\"path\" string=\"true\">note.txt</||DSML|| parameter>\n</||DSML|| invoke>\n</||DSML|| calls>\n"; |
| 1215 | assert!(contains_fake_tool_wrapper(text)); |
| 1216 | for cut in 1..text.len() { |
| 1217 | if !text.is_char_boundary(cut) { |
| 1218 | continue; |
| 1219 | } |
| 1220 | let mut state = ToolCallDeltaFilterState::default(); |
| 1221 | let mut visible = filter_tool_call_delta_with_state(&text[..cut], &mut state); |
| 1222 | visible.push_str(&filter_tool_call_delta_with_state(&text[cut..], &mut state)); |
| 1223 | visible.push_str(&flush_tool_call_delta_state(&mut state)); |
| 1224 | assert_eq!(visible.trim(), "", "cut {cut} leaked DSML: {visible:?}"); |
| 1225 | } |
| 1226 | } |
| 1227 | // from filter_tool_call_delta_strips_deepseek_native_token_split_across_chunks |
| 1228 | { |
| 1229 | // The streaming filter carries a partial marker across chunk boundaries. |
| 1230 | // These markers are multi-byte, so a split partway through one is the case |
| 1231 | // most likely to slip past the carry buffer. |
| 1232 | let mut state = ToolCallDeltaFilterState::default(); |
| 1233 | let full = "before <|tool▁calls▁begin|>payload<|tool▁calls▁end|> after"; |
| 1234 | let cut = full.find("calls").expect("marker present") + 2; |
| 1235 | let mut visible = filter_tool_call_delta_with_state(&full[..cut], &mut state); |
| 1236 | visible.push_str(&filter_tool_call_delta_with_state(&full[cut..], &mut state)); |
| 1237 | |
| 1238 | assert!( |
| 1239 | !visible.contains("tool▁") && !visible.contains("payload"), |
| 1240 | "chunk-split marker leaked: {visible:?}" |
| 1241 | ); |
| 1242 | assert!(visible.contains("before"), "{visible:?}"); |
| 1243 | assert!(visible.contains("after"), "{visible:?}"); |
| 1244 | } |
| 1245 | // from filter_tool_call_delta_strips_generic_tool_call_marker |
| 1246 | { |
| 1247 | let mut in_block = false; |
| 1248 | let visible = filter_tool_call_delta( |
| 1249 | "lead <tool_call>\n{\"name\":\"do\"}\n</tool_call> tail", |
| 1250 | &mut in_block, |
| 1251 | ); |
| 1252 | assert!(!in_block); |
| 1253 | assert!(!visible.contains("<tool_call")); |
| 1254 | assert!(!visible.contains("</tool_call>")); |
| 1255 | assert!(visible.contains("lead")); |
| 1256 | assert!(visible.contains("tail")); |
| 1257 | } |
| 1258 | // from filter_tool_call_delta_strips_invoke_marker |
| 1259 | { |
| 1260 | let mut in_block = false; |
| 1261 | let visible = filter_tool_call_delta( |
| 1262 | "alpha <invoke name=\"x\"><parameter name=\"k\">v</parameter></invoke> beta", |
| 1263 | &mut in_block, |
| 1264 | ); |
| 1265 | assert!(!in_block); |
| 1266 | assert!(!visible.contains("<invoke ")); |
| 1267 | assert!(!visible.contains("</invoke>")); |
| 1268 | assert!(visible.contains("alpha")); |
| 1269 | assert!(visible.contains("beta")); |
| 1270 | } |
| 1271 | // from filter_tool_call_delta_strips_function_calls_marker |
| 1272 | { |
| 1273 | let mut in_block = false; |
| 1274 | let visible = filter_tool_call_delta( |
| 1275 | "head <function_calls>\n{\"name\":\"x\"}\n</function_calls> tail", |
| 1276 | &mut in_block, |
| 1277 | ); |
| 1278 | assert!(!in_block); |
| 1279 | assert!(!visible.contains("<function_calls>")); |
| 1280 | assert!(!visible.contains("</function_calls>")); |
| 1281 | assert!(visible.contains("head")); |
| 1282 | assert!(visible.contains("tail")); |
| 1283 | } |
| 1284 | // from filter_tool_call_delta_strips_siliconflow_v4_dsml_content_fixture |
| 1285 | { |
| 1286 | // #2900: a SiliconFlow CN `deepseek-ai/DeepSeek-V4-Pro` stream can leak |
| 1287 | // DSML/function-call markup through the ordinary content channel. Keep it |
| 1288 | // out of visible assistant text; do not reinterpret `<function_calls>` as |
| 1289 | // an executable legacy text tool call. |
| 1290 | let mut in_block = false; |
| 1291 | let visible_a = filter_tool_call_delta( |
| 1292 | "visible prefix <function_calls>\n{\"name\":\"exec_shell\",\"arguments\":{\"cmd\":\"echo leaked\"}}", |
| 1293 | &mut in_block, |
| 1294 | ); |
| 1295 | assert!(in_block); |
| 1296 | assert_eq!(visible_a, "visible prefix "); |
| 1297 | |
| 1298 | let visible_b = filter_tool_call_delta("\n</function_calls> visible suffix", &mut in_block); |
| 1299 | assert!(!in_block); |
| 1300 | assert_eq!(visible_b, " visible suffix"); |
| 1301 | assert!(!visible_b.contains("exec_shell")); |
| 1302 | assert!(!visible_b.contains("<function_calls>")); |
| 1303 | } |
| 1304 | } |
| 1305 | |
| 1306 | #[test] |
| 1307 | fn marker_tables_are_consistent() { |
| 1308 | // Three parallel tables describe the same wrapper shapes. They drifting |
| 1309 | // apart is exactly how #3880's family went unhandled, so assert they |
| 1310 | // agree rather than trusting review. |
| 1311 | assert_eq!( |
| 1312 | TOOL_CALL_MARKER_PAIRS.len(), |
| 1313 | TOOL_CALL_START_MARKERS.len(), |
| 1314 | "start-marker table is out of sync with the pair table" |
| 1315 | ); |
| 1316 | assert_eq!( |
| 1317 | TOOL_CALL_MARKER_PAIRS.len(), |
| 1318 | TOOL_CALL_END_MARKERS.len(), |
| 1319 | "end-marker table is out of sync with the pair table" |
| 1320 | ); |
| 1321 | for (index, (start, end)) in TOOL_CALL_MARKER_PAIRS.iter().enumerate() { |
| 1322 | assert_eq!( |
| 1323 | *start, TOOL_CALL_START_MARKERS[index], |
| 1324 | "start marker {index} disagrees with the pair table" |
| 1325 | ); |
| 1326 | assert_eq!( |
| 1327 | *end, TOOL_CALL_END_MARKERS[index], |
| 1328 | "end marker {index} disagrees with the pair table" |
| 1329 | ); |
| 1330 | } |
| 1331 | } |
| 1332 | |
| 1333 | #[test] |
| 1334 | fn filter_tool_scenario_2() { |
| 1335 | // Scenario consolidation of: filter_tool_call_delta_strips_fullwidth_dsml_invoke_fixture, filter_tool_call_delta_strips_ascii_dsml_invoke_fixture, filter_tool_call_delta_carries_split_fullwidth_dsml_marker, filter_tool_call_delta_flushes_clean_partial_marker_prefix, filter_tool_call_delta_handles_chunk_split_marker, filter_tool_call_delta_unmatched_open_suppresses_remainder, filter_tool_call_delta_passes_through_clean_text |
| 1336 | // from filter_tool_call_delta_strips_fullwidth_dsml_invoke_fixture |
| 1337 | { |
| 1338 | // #3717: Windows users reported SiliconFlow/DSML content leaking through |
| 1339 | // the ordinary text channel with fullwidth DSML wrapper tags. Treat it as |
| 1340 | // non-API tool markup, not visible assistant text. |
| 1341 | let mut in_block = false; |
| 1342 | let visible = filter_tool_call_delta( |
| 1343 | "visible prefix <|DSML|tool_calls>\n\ |
| 1344 | <|DSML|invoke name=\"read_file\">\n\ |
| 1345 | <|DSML|parameter name=\"path\" string=\"true\">backend/open_webui/utils/auth.py</|DSML|parameter>\n\ |
| 1346 | </|DSML|invoke>\n\ |
| 1347 | </|DSML|tool_calls> visible suffix", |
| 1348 | &mut in_block, |
| 1349 | ); |
| 1350 | |
| 1351 | assert!(!in_block); |
| 1352 | assert_eq!(visible, "visible prefix visible suffix"); |
| 1353 | assert!(!visible.contains("DSML")); |
| 1354 | assert!(!visible.contains("read_file")); |
| 1355 | assert!(!visible.contains("backend/open_webui")); |
| 1356 | } |
| 1357 | // from filter_tool_call_delta_strips_ascii_dsml_invoke_fixture |
| 1358 | { |
| 1359 | let mut in_block = false; |
| 1360 | let visible = filter_tool_call_delta( |
| 1361 | "visible prefix <|DSML|tool_calls>\n\ |
| 1362 | <|DSML|invoke name=\"read_file\">\n\ |
| 1363 | <|DSML|parameter name=\"path\" string=\"true\">backend/open_webui/utils/auth.py</|DSML|parameter>\n\ |
| 1364 | </|DSML|invoke>\n\ |
| 1365 | </|DSML|tool_calls> visible suffix", |
| 1366 | &mut in_block, |
| 1367 | ); |
| 1368 | |
| 1369 | assert!(!in_block); |
| 1370 | assert_eq!(visible, "visible prefix visible suffix"); |
| 1371 | assert!(!visible.contains("DSML")); |
| 1372 | assert!(!visible.contains("read_file")); |
| 1373 | assert!(!visible.contains("backend/open_webui")); |
| 1374 | } |
| 1375 | // from filter_tool_call_delta_carries_split_fullwidth_dsml_marker |
| 1376 | { |
| 1377 | let mut state = ToolCallDeltaFilterState::default(); |
| 1378 | |
| 1379 | let visible_a = filter_tool_call_delta_with_state("visible prefix <|DS", &mut state); |
| 1380 | assert_eq!(visible_a, "visible prefix "); |
| 1381 | |
| 1382 | let visible_b = filter_tool_call_delta_with_state( |
| 1383 | "ML|tool_calls>\n<|DSML|invoke name=\"read_file\">", |
| 1384 | &mut state, |
| 1385 | ); |
| 1386 | assert!( |
| 1387 | visible_b.is_empty(), |
| 1388 | "split DSML opener leaked: {visible_b:?}" |
| 1389 | ); |
| 1390 | |
| 1391 | let visible_c = filter_tool_call_delta_with_state( |
| 1392 | "</|DSML|invoke>\n</|DSML|tool_calls> visible suffix", |
| 1393 | &mut state, |
| 1394 | ); |
| 1395 | assert_eq!(visible_c, " visible suffix"); |
| 1396 | } |
| 1397 | // from filter_tool_call_delta_flushes_clean_partial_marker_prefix |
| 1398 | { |
| 1399 | let mut state = ToolCallDeltaFilterState::default(); |
| 1400 | |
| 1401 | let visible = filter_tool_call_delta_with_state("ordinary text ending in <", &mut state); |
| 1402 | assert_eq!(visible, "ordinary text ending in "); |
| 1403 | |
| 1404 | let flushed = flush_tool_call_delta_state(&mut state); |
| 1405 | assert_eq!(flushed, "<"); |
| 1406 | } |
| 1407 | // from filter_tool_call_delta_handles_chunk_split_marker |
| 1408 | { |
| 1409 | let mut in_block = false; |
| 1410 | // First chunk opens the wrapper but does not close it. |
| 1411 | let visible_a = filter_tool_call_delta("hello <tool_call>partial", &mut in_block); |
| 1412 | assert!(in_block, "filter must remember it is mid-wrapper"); |
| 1413 | assert_eq!(visible_a, "hello "); |
| 1414 | |
| 1415 | // Second chunk continues inside the wrapper, then closes it and adds tail. |
| 1416 | let visible_b = filter_tool_call_delta("payload</tool_call> tail", &mut in_block); |
| 1417 | assert!(!in_block); |
| 1418 | assert_eq!(visible_b, " tail"); |
| 1419 | } |
| 1420 | // from filter_tool_call_delta_unmatched_open_suppresses_remainder |
| 1421 | { |
| 1422 | let mut in_block = false; |
| 1423 | let visible = filter_tool_call_delta("ok [TOOL_CALL]rest of stream", &mut in_block); |
| 1424 | assert_eq!(visible, "ok "); |
| 1425 | assert!( |
| 1426 | in_block, |
| 1427 | "unmatched open must leave filter in tool-call mode" |
| 1428 | ); |
| 1429 | } |
| 1430 | // from filter_tool_call_delta_passes_through_clean_text |
| 1431 | { |
| 1432 | let mut in_block = false; |
| 1433 | let input = "no markers here, just prose with code `<not a tag>`."; |
| 1434 | let visible = filter_tool_call_delta(input, &mut in_block); |
| 1435 | assert!(!in_block); |
| 1436 | assert_eq!(visible, input); |
| 1437 | } |
| 1438 | } |
| 1439 | |
| 1440 | #[test] |
| 1441 | fn contains_fake_scenario() { |
| 1442 | // Scenario consolidation of: contains_fake_tool_wrapper_detects_each_marker, contains_fake_tool_wrapper_returns_false_on_clean_text |
| 1443 | // from contains_fake_tool_wrapper_detects_each_marker |
| 1444 | { |
| 1445 | for marker in TOOL_CALL_START_MARKERS { |
| 1446 | let needle = format!("noise {marker} more noise"); |
| 1447 | assert!( |
| 1448 | contains_fake_tool_wrapper(&needle), |
| 1449 | "marker `{marker}` should be detected" |
| 1450 | ); |
| 1451 | } |
| 1452 | } |
| 1453 | // from contains_fake_tool_wrapper_returns_false_on_clean_text |
| 1454 | { |
| 1455 | assert!(!contains_fake_tool_wrapper( |
| 1456 | "plain assistant text without wrappers" |
| 1457 | )); |
| 1458 | assert!(!contains_fake_tool_wrapper( |
| 1459 | "`<tool` lookalike but not a real start marker" |
| 1460 | )); |
| 1461 | } |
| 1462 | } |
| 1463 | |
| 1464 | #[test] |
| 1465 | fn fake_wrapper_notice_is_compact_and_actionable() { |
| 1466 | // Keep this short so it fits cleanly in a single status line. |
| 1467 | assert!(FAKE_WRAPPER_NOTICE.len() < 120); |
| 1468 | assert!(FAKE_WRAPPER_NOTICE.contains("API tool channel")); |
| 1469 | } |
| 1470 | |
| 1471 | // ---- final_tool_input: bug-class regression for "<command>" placeholder ---- |
| 1472 | // |
| 1473 | // Background: a streamed tool block carries its `input` in two pieces — an |
| 1474 | // initial value at `ContentBlockStart` (often `{}`), then `InputJsonDelta` |
| 1475 | // chunks that build up `input_buffer`. The TUI used to fire `ToolCallStarted` |
| 1476 | // from `ContentBlockStart` with the empty initial input and never re-emit |
| 1477 | // once args were known, so cells rendered the literal text `<command>` / |
| 1478 | // `<file>` placeholders. The input is finalized at `ContentBlockStop`, then published after batch admission |
| 1479 | // through `final_tool_input`, which prefers the parsed |
| 1480 | // buffer over a stale empty placeholder. |
| 1481 | fn tool_state(initial: serde_json::Value, buffer: &str) -> ToolUseState { |
| 1482 | ToolUseState { |
| 1483 | execution_id: uuid::Uuid::new_v4().to_string(), |
| 1484 | id: "t1".into(), |
| 1485 | name: "exec_shell".into(), |
| 1486 | input: initial, |
| 1487 | caller: None, |
| 1488 | thought_signature: None, |
| 1489 | input_buffer: buffer.into(), |
| 1490 | input_parse_error: None, |
| 1491 | } |
| 1492 | } |
| 1493 | |
| 1494 | #[test] |
| 1495 | fn final_tool_scenario() { |
| 1496 | // Scenario consolidation of: final_tool_input_prefers_parsed_buffer_over_empty_initial, final_tool_input_falls_back_to_initial_when_buffer_empty, final_tool_input_preserves_raw_buffer_for_parse_errors |
| 1497 | // from final_tool_input_prefers_parsed_buffer_over_empty_initial |
| 1498 | { |
| 1499 | // The exact regression: ContentBlockStart delivered `{}`, then args |
| 1500 | // streamed in via InputJsonDelta. The emitted ToolCallStarted must |
| 1501 | // carry the parsed buffer, not the placeholder. |
| 1502 | let state = tool_state(json!({}), r#"{"command": "ls -la"}"#); |
| 1503 | assert_eq!(final_tool_input(&state), json!({"command": "ls -la"})); |
| 1504 | } |
| 1505 | // from final_tool_input_falls_back_to_initial_when_buffer_empty |
| 1506 | { |
| 1507 | // Models occasionally embed args directly in the start frame and never |
| 1508 | // send any InputJsonDelta. We must still report those args. |
| 1509 | let state = tool_state(json!({"command": "echo hi"}), ""); |
| 1510 | assert_eq!(final_tool_input(&state), json!({"command": "echo hi"})); |
| 1511 | } |
| 1512 | // from final_tool_input_preserves_raw_buffer_for_parse_errors |
| 1513 | { |
| 1514 | let mut state = tool_state(json!({}), "{not json"); |
| 1515 | state.input_parse_error = Some("malformed tool arguments".into()); |
| 1516 | assert_eq!( |
| 1517 | final_tool_input(&state), |
| 1518 | json!({"raw_arguments": "{not json"}) |
| 1519 | ); |
| 1520 | } |
| 1521 | // A `write` whose stream was cut at its output limit, right after a |
| 1522 | // complete string value. `arg_repair` CAN make this parse by appending |
| 1523 | // one `}`, and before the repair ladder reported provenance that guess |
| 1524 | // was dispatched — writing a file containing only "first line" while the |
| 1525 | // model was still mid-argument. It must now take the malformed path, so |
| 1526 | // the model is told to re-issue instead. |
| 1527 | { |
| 1528 | let state = tool_state(json!({}), r#"{"path": "notes.md", "content": "first line""#); |
| 1529 | assert_eq!( |
| 1530 | final_tool_input(&state), |
| 1531 | json!({"raw_arguments": r#"{"path": "notes.md", "content": "first line""#}), |
| 1532 | "a truncated write must not be dispatched as a completed argument" |
| 1533 | ); |
| 1534 | } |
| 1535 | // The guard must not fire on arguments that were merely sloppy: a |
| 1536 | // trailing comma is structurally complete and still has to dispatch, or |
| 1537 | // every DeepSeek chunk-boundary repair would start failing tool calls. |
| 1538 | { |
| 1539 | let state = tool_state(json!({}), r#"{"command": "ls -la",}"#); |
| 1540 | assert_eq!(final_tool_input(&state), json!({"command": "ls -la"})); |
| 1541 | } |
| 1542 | } |
| 1543 | |
| 1544 | // === #103 transparent stream-retry policy ===================================== |
| 1545 | |
| 1546 | #[test] |
| 1547 | fn stream_retry_scenario() { |
| 1548 | // Scenario consolidation of: stream_retry_zero_content_then_error_is_transparently_retried, stream_retry_after_content_received_surfaces_error, stream_retry_respects_cancellation, stream_retry_budget_caps_resumes_in_mechanism, stream_retry_threshold_relaxed_to_five |
| 1549 | // from stream_retry_zero_content_then_error_is_transparently_retried |
| 1550 | { |
| 1551 | // Case 2 from issue #103: stream yielded ZERO content then errored. |
| 1552 | // The decoder hit Err on the very first poll → engine should retry |
| 1553 | // because DeepSeek hasn't billed and the user has seen nothing. |
| 1554 | assert!( |
| 1555 | super::should_transparently_retry_stream( |
| 1556 | false, |
| 1557 | 0, |
| 1558 | super::MAX_TRANSPARENT_STREAM_RETRIES, |
| 1559 | false |
| 1560 | ), |
| 1561 | "first attempt with no content must be eligible for transparent retry" |
| 1562 | ); |
| 1563 | assert!( |
| 1564 | super::should_transparently_retry_stream( |
| 1565 | false, |
| 1566 | 1, |
| 1567 | super::MAX_TRANSPARENT_STREAM_RETRIES, |
| 1568 | false |
| 1569 | ), |
| 1570 | "second attempt (one prior retry) with no content must still be eligible" |
| 1571 | ); |
| 1572 | } |
| 1573 | // from stream_retry_after_content_received_surfaces_error |
| 1574 | { |
| 1575 | // Case 3 from issue #103: stream yielded content then errored. We must |
| 1576 | // NOT transparently retry — the model has emitted billed output tokens |
| 1577 | // and the UI has streamed deltas; resending would double-bill and the |
| 1578 | // user would see the same prefix twice. |
| 1579 | assert!( |
| 1580 | !super::should_transparently_retry_stream( |
| 1581 | true, |
| 1582 | 0, |
| 1583 | super::MAX_TRANSPARENT_STREAM_RETRIES, |
| 1584 | false |
| 1585 | ), |
| 1586 | "any content received → no transparent retry, even with full budget" |
| 1587 | ); |
| 1588 | assert!( |
| 1589 | !super::should_transparently_retry_stream( |
| 1590 | true, |
| 1591 | 1, |
| 1592 | super::MAX_TRANSPARENT_STREAM_RETRIES, |
| 1593 | false |
| 1594 | ), |
| 1595 | "any content received → no transparent retry on subsequent attempts" |
| 1596 | ); |
| 1597 | } |
| 1598 | // from stream_retry_respects_cancellation |
| 1599 | { |
| 1600 | // Cancellation overrides every other condition. If the user pressed |
| 1601 | // Esc / Ctrl-C, do not silently re-issue the request behind their back. |
| 1602 | assert!( |
| 1603 | !super::should_transparently_retry_stream( |
| 1604 | false, |
| 1605 | 0, |
| 1606 | super::MAX_TRANSPARENT_STREAM_RETRIES, |
| 1607 | true |
| 1608 | ), |
| 1609 | "cancelled turn must not be transparently retried" |
| 1610 | ); |
| 1611 | assert!( |
| 1612 | !super::should_transparently_retry_stream( |
| 1613 | false, |
| 1614 | 1, |
| 1615 | super::MAX_TRANSPARENT_STREAM_RETRIES, |
| 1616 | true |
| 1617 | ), |
| 1618 | "cancelled turn must not be transparently retried even with budget" |
| 1619 | ); |
| 1620 | } |
| 1621 | // from stream_retry_budget_caps_resumes_in_mechanism |
| 1622 | { |
| 1623 | // "At most one bounded retry per drop" is enforced by types, not by a |
| 1624 | // comment: `authorize()` is the only way to spend a resume and it refuses |
| 1625 | // once MAX_STREAM_RETRIES resumes have been issued, whatever the guard |
| 1626 | // predicates say. A healthy round resets the chain. |
| 1627 | let mut budget = super::StreamRetryBudget::default(); |
| 1628 | assert_eq!(budget.spent(), 0); |
| 1629 | assert_eq!(budget.authorize(), Some(1)); |
| 1630 | assert_eq!(budget.authorize(), Some(2)); |
| 1631 | assert_eq!(budget.authorize(), Some(3)); |
| 1632 | assert_eq!( |
| 1633 | budget.authorize(), |
| 1634 | None, |
| 1635 | "authorize() must refuse past MAX_STREAM_RETRIES" |
| 1636 | ); |
| 1637 | assert_eq!(budget.authorize(), None, "and keep refusing"); |
| 1638 | assert_eq!(budget.spent(), super::MAX_STREAM_RETRIES); |
| 1639 | budget.reset(); |
| 1640 | assert_eq!(budget.spent(), 0); |
| 1641 | assert_eq!(budget.authorize(), Some(1)); |
| 1642 | } |
| 1643 | // from stream_retry_threshold_relaxed_to_five |
| 1644 | { |
| 1645 | // Case 1+4 from issue #103: the consecutive-error threshold for marking |
| 1646 | // the turn failed was relaxed from 3 → 5 in v0.6.7 because the new |
| 1647 | // HTTP/2 keepalive defaults make spurious decode errors rarer. |
| 1648 | // This test pins the constant so a future regression to 3 fails loudly. |
| 1649 | assert_eq!( |
| 1650 | super::MAX_STREAM_ERRORS_BEFORE_FAIL, |
| 1651 | 5, |
| 1652 | "the consecutive-stream-error threshold should be 5; \ |
| 1653 | lowering it back to 3 will fail mid-turn under transient flakiness" |
| 1654 | ); |
| 1655 | // And a regression guard on the transparent-retry cap. |
| 1656 | assert_eq!( |
| 1657 | super::MAX_TRANSPARENT_STREAM_RETRIES, |
| 1658 | 2, |
| 1659 | "transparent-retry cap should be 2; raising it risks hammering the \ |
| 1660 | provider on real outages" |
| 1661 | ); |
| 1662 | } |
| 1663 | } |
| 1664 |