返回 CodeWhale
test_cases_12.rs
根目录 / crates / tui / src / core / engine / tests / test_cases_12.rs
1
2
3 #[tokio::test]
4 #[allow(clippy::await_holding_lock)]
5 async fn auto_review_asks_the_user_and_returns_the_answer() {
6 use wiremock::matchers::{body_string_contains, method, path};
7 use wiremock::{Mock, MockServer, ResponseTemplate};
8
9 let _lock = lock_test_env();
10 let workspace = tempdir().expect("tempdir");
11 let server = MockServer::start().await;
12 let arguments = json!({
13 "questions": [{
14 "header": "Choice",
15 "id": "choice",
16 "question": "Which path should I take?",
17 "options": [
18 {"label": "A", "description": "Take path A"},
19 {"label": "B", "description": "Take path B"}
20 ]
21 }]
22 })
23 .to_string();
24 let tool_delta = json!({
25 "id": "chatcmpl-auto-review-question",
26 "choices": [{
27 "index": 0,
28 "delta": {"tool_calls": [{
29 "index": 0,
30 "id": "call_auto_review_question",
31 "type": "function",
32 "function": {
33 "name": REQUEST_USER_INPUT_NAME,
34 "arguments": arguments,
35 },
36 }]},
37 "finish_reason": serde_json::Value::Null,
38 }],
39 });
40 let tool_finish = json!({
41 "id": "chatcmpl-auto-review-question",
42 "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
43 });
44 let tool_call_sse = format!("data: {tool_delta}\n\ndata: {tool_finish}\n\ndata: [DONE]\n\n");
45 let done_sse = concat!(
46 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
47 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
48 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
49 "\"finish_reason\":\"stop\"}]}\n\n",
50 "data: [DONE]\n\n",
51 );
52
53 Mock::given(method("POST"))
54 .and(path("/v1/chat/completions"))
55 .and(body_string_contains("take-path-b"))
56 .respond_with(
57 ResponseTemplate::new(200)
58 .insert_header("content-type", "text/event-stream")
59 .set_body_string(done_sse),
60 )
61 .expect(1)
62 .with_priority(1)
63 .mount(&server)
64 .await;
65 Mock::given(method("POST"))
66 .and(path("/v1/chat/completions"))
67 .respond_with(
68 ResponseTemplate::new(200)
69 .insert_header("content-type", "text/event-stream")
70 .set_body_string(tool_call_sse),
71 )
72 .expect(1)
73 .with_priority(2)
74 .mount(&server)
75 .await;
76
77 let api_config = Config {
78 ..Config::default()
79 }
80 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
81 let (engine, handle) = Engine::new(
82 EngineConfig {
83 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
84 workspace: workspace.path().to_path_buf(),
85 snapshots_enabled: false,
86 subagents_enabled: false,
87 ..EngineConfig::default()
88 },
89 &api_config,
90 );
91 let run_task = tokio::spawn(engine.run());
92
93 handle
94 .send(Op::SendMessage(TurnSpec {
95 max_output_tokens: None,
96 content: "continue autonomously".to_string(),
97 images: Vec::new(),
98 mode: AppMode::Agent,
99 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
100 compaction: Box::new(CompactionConfig::default()),
101 initial_routed_usage: Box::default(),
102 goal_objective: None,
103 goal_token_budget: None,
104 goal_status: crate::tools::goal::GoalStatus::Active,
105 reasoning_effort: None,
106 reasoning_effort_auto: false,
107 auto_model: false,
108 allow_shell: true,
109 trust_mode: false,
110 auto_approve: false,
111 approval_mode: ApprovalMode::Auto,
112 translation_enabled: false,
113 allowed_tools: None,
114 dynamic_tools: Vec::new(),
115 hook_executor: None,
116 verbosity: None,
117 provenance: UserInputProvenance::ExternalUser,
118 submission_id: None,
119 }))
120 .await
121 .expect("send Auto-Review model turn");
122
123 let mut saw_question = false;
124 let mut saw_tool_result = false;
125 let mut saw_turn_complete = false;
126 let mut rx = handle.rx_event.write().await;
127 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
128 .await
129 .expect("timed out waiting for Auto-Review question event")
130 {
131 match event {
132 Event::UserInputRequired { id, .. } => {
133 // Auto-Review reviews tool approvals only; a question still
134 // reaches the user.
135 handle
136 .submit_user_input(
137 id,
138 crate::tools::user_input::UserInputResponse {
139 answers: vec![crate::tools::user_input::UserInputAnswer {
140 id: "choice".to_string(),
141 label: "B".to_string(),
142 value: "take-path-b".to_string(),
143 }],
144 },
145 )
146 .await
147 .expect("submit answer");
148 saw_question = true;
149 }
150 Event::ApprovalRequired { .. } => {
151 panic!("an Auto-Review question must not become an approval")
152 }
153 Event::ToolCallComplete { name, result, .. } if name == REQUEST_USER_INPUT_NAME => {
154 let result = result.expect("the answered question succeeds");
155 assert!(result.success, "{result:?}");
156 assert!(result.content.contains("take-path-b"), "{result:?}");
157 saw_tool_result = true;
158 }
159 Event::TurnComplete { status, .. } => {
160 assert_eq!(status, TurnOutcomeStatus::Completed);
161 saw_turn_complete = true;
162 break;
163 }
164 _ => {}
165 }
166 }
167 drop(rx);
168
169 handle.send(Op::Shutdown).await.expect("shutdown engine");
170 run_task.await.expect("engine task");
171 assert!(saw_question);
172 assert!(saw_tool_result);
173 assert!(saw_turn_complete);
174 }
175
176 #[tokio::test]
177 #[allow(clippy::await_holding_lock)]
178 async fn full_access_permission_allow_cannot_bypass_background_catastrophic_floor() {
179 use wiremock::matchers::{body_string_contains, method, path};
180 use wiremock::{Mock, MockServer, ResponseTemplate};
181
182 let _lock = lock_test_env();
183 let workspace = tempdir().expect("tempdir");
184 let server = MockServer::start().await;
185 let victim = workspace.path().join("must-survive");
186 fs::write(&victim, "guarded\n").expect("write guarded fixture");
187 // Keep the engine-boundary regression intrinsically harmless on every
188 // runner: the quoted payload trips the same built-in catastrophic-command
189 // detector, while an execution regression would only overwrite the
190 // sentinel. The policy-level sibling tests exercise real destructive
191 // command shapes directly without ever dispatching them to a shell.
192 let command = format!("echo \"rm -rf /\" > \"{}\"", victim.display());
193 let allow_rule = codewhale_execpolicy::ToolAskRule::exec_shell(command.clone())
194 .into_exact_workspace_allow(workspace.path().to_string_lossy().into_owned());
195 let tool_input = json!({
196 "action": "run",
197 "command": command,
198 "background": true,
199 });
200 let arguments = serde_json::to_string(&tool_input).expect("serialize tool arguments");
201
202 let tool_delta = json!({
203 "id": "chatcmpl-bg",
204 "choices": [{
205 "index": 0,
206 "delta": {"tool_calls": [{
207 "index": 0,
208 "id": "call_bg",
209 "type": "function",
210 "function": {"name": "Bash", "arguments": arguments},
211 }]},
212 "finish_reason": serde_json::Value::Null,
213 }],
214 });
215 let tool_finish = json!({
216 "id": "chatcmpl-bg",
217 "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
218 });
219 let tool_call_sse = format!("data: {tool_delta}\n\ndata: {tool_finish}\n\ndata: [DONE]\n\n");
220 let done_sse = concat!(
221 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
222 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
223 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
224 "\"finish_reason\":\"stop\"}]}\n\n",
225 "data: [DONE]\n\n",
226 );
227
228 Mock::given(method("POST"))
229 .and(path("/v1/chat/completions"))
230 .and(body_string_contains("destructive background/headless"))
231 .respond_with(
232 ResponseTemplate::new(200)
233 .insert_header("content-type", "text/event-stream")
234 .set_body_string(done_sse),
235 )
236 .expect(1)
237 .with_priority(1)
238 .mount(&server)
239 .await;
240 Mock::given(method("POST"))
241 .and(path("/v1/chat/completions"))
242 .respond_with(
243 ResponseTemplate::new(200)
244 .insert_header("content-type", "text/event-stream")
245 .set_body_string(tool_call_sse),
246 )
247 .expect(1)
248 .with_priority(2)
249 .mount(&server)
250 .await;
251
252 let api_config = Config {
253 ..Config::default()
254 }
255 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
256 let engine_config = EngineConfig {
257 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
258 workspace: workspace.path().to_path_buf(),
259 snapshots_enabled: false,
260 subagents_enabled: false,
261 exec_policy_engine: codewhale_execpolicy::ExecPolicyEngine::with_rulesets(vec![
262 codewhale_execpolicy::Ruleset::user(vec![], vec![]).with_ask_rules(vec![allow_rule]),
263 ]),
264 ..EngineConfig::default()
265 };
266 assert_eq!(
267 exec_shell_ask_rule_decision(
268 &engine_config,
269 "exec_shell",
270 &tool_input,
271 workspace.path(),
272 ApprovalMode::Bypass,
273 ),
274 Some(ToolAskRuleDecision::Allow),
275 "precondition: the remembered grant must match before the safety floor tightens the plan"
276 );
277 let (engine, handle) = Engine::new(engine_config, &api_config);
278 let run_task = tokio::spawn(engine.run());
279
280 handle
281 .send(Op::SendMessage(TurnSpec {
282 max_output_tokens: None,
283 content: "please run a background shell".to_string(),
284 images: Vec::new(),
285 mode: AppMode::Agent,
286 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
287 compaction: Box::new(CompactionConfig::default()),
288 initial_routed_usage: Box::default(),
289 goal_objective: None,
290 goal_token_budget: None,
291 goal_status: crate::tools::goal::GoalStatus::Active,
292 reasoning_effort: None,
293 reasoning_effort_auto: false,
294 auto_model: false,
295 allow_shell: true,
296 trust_mode: true,
297 auto_approve: true,
298 approval_mode: ApprovalMode::Bypass,
299 translation_enabled: false,
300 allowed_tools: None,
301 dynamic_tools: Vec::new(),
302 hook_executor: None,
303 verbosity: None,
304 provenance: UserInputProvenance::ExternalUser,
305 submission_id: None,
306 }))
307 .await
308 .expect("send model turn");
309
310 let mut saw_tool_result = false;
311 let mut saw_complete = false;
312 let mut rx = handle.rx_event.write().await;
313 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
314 .await
315 .expect("timed out waiting for engine event")
316 {
317 match event {
318 Event::ApprovalRequired { .. } => {
319 panic!("Full Access safety holds must fail closed without prompting")
320 }
321 Event::ToolCallComplete { name, result, .. } => {
322 if name == "Bash" {
323 saw_tool_result = true;
324 let err = result.expect_err("blocked shell should not execute");
325 assert!(
326 err.to_string().contains("Built-in safety gate"),
327 "unexpected shell denial: {err:?}"
328 );
329 }
330 }
331 Event::TurnComplete { status, .. } => {
332 assert_eq!(status, TurnOutcomeStatus::Completed);
333 saw_complete = true;
334 break;
335 }
336 _ => {}
337 }
338 }
339 drop(rx);
340
341 handle.send(Op::Shutdown).await.expect("shutdown engine");
342 run_task.await.expect("engine task");
343 assert!(saw_tool_result);
344 assert!(saw_complete);
345 assert_eq!(
346 fs::read_to_string(&victim).expect("read guarded fixture"),
347 "guarded\n",
348 "blocked command must not touch its target"
349 );
350 }
351
352 #[tokio::test]
353 #[allow(clippy::await_holding_lock)]
354 async fn yolo_mode_does_not_prompt_for_background_shell() {
355 // #3883: the durable-review floor keys on what the command does, not on
356 // "not provably read-only". An ordinary background command in YOLO must
357 // run without a prompt; genuinely destructive and publish-like background
358 // work still holds (see the sibling tests).
359 use wiremock::matchers::{body_string_contains, method, path};
360 use wiremock::{Mock, MockServer, ResponseTemplate};
361
362 let _lock = lock_test_env();
363 let workspace = tempdir().expect("tempdir");
364 let server = MockServer::start().await;
365
366 let tool_call_sse = concat!(
367 "data: {\"id\":\"chatcmpl-bgok\",\"choices\":[{\"index\":0,\"delta\":{\"tool_calls\":[",
368 "{\"index\":0,\"id\":\"call_bgok\",\"type\":\"function\",\"function\":{\"name\":\"Bash\",",
369 "\"arguments\":\"{\\\"action\\\":\\\"run\\\",\\\"command\\\":\\\"echo bg-yolo-no-prompt\\\",\\\"background\\\":true}\"}}",
370 "]},\"finish_reason\":null}]}\n\n",
371 "data: {\"id\":\"chatcmpl-bgok\",\"choices\":[{\"index\":0,\"delta\":{},",
372 "\"finish_reason\":\"tool_calls\"}]}\n\n",
373 "data: [DONE]\n\n",
374 );
375 let done_sse = concat!(
376 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
377 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
378 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
379 "\"finish_reason\":\"stop\"}]}\n\n",
380 "data: [DONE]\n\n",
381 );
382
383 Mock::given(method("POST"))
384 .and(path("/v1/chat/completions"))
385 .and(body_string_contains("bg-yolo-no-prompt"))
386 .respond_with(
387 ResponseTemplate::new(200)
388 .insert_header("content-type", "text/event-stream")
389 .set_body_string(done_sse),
390 )
391 .expect(1)
392 .with_priority(1)
393 .mount(&server)
394 .await;
395 Mock::given(method("POST"))
396 .and(path("/v1/chat/completions"))
397 .respond_with(
398 ResponseTemplate::new(200)
399 .insert_header("content-type", "text/event-stream")
400 .set_body_string(tool_call_sse),
401 )
402 .expect(1)
403 .with_priority(2)
404 .mount(&server)
405 .await;
406
407 let api_config = Config {
408 ..Config::default()
409 }
410 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
411 let (engine, handle) = Engine::new(
412 EngineConfig {
413 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
414 workspace: workspace.path().to_path_buf(),
415 snapshots_enabled: false,
416 subagents_enabled: false,
417 ..EngineConfig::default()
418 },
419 &api_config,
420 );
421 let run_task = tokio::spawn(engine.run());
422
423 handle
424 .send(Op::SendMessage(TurnSpec {
425 max_output_tokens: None,
426 content: "please run a background shell".to_string(),
427 images: Vec::new(),
428 mode: AppMode::Agent,
429 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
430 compaction: Box::new(CompactionConfig::default()),
431 initial_routed_usage: Box::default(),
432 goal_objective: None,
433 goal_token_budget: None,
434 goal_status: crate::tools::goal::GoalStatus::Active,
435 reasoning_effort: None,
436 reasoning_effort_auto: false,
437 auto_model: false,
438 allow_shell: true,
439 trust_mode: true,
440 auto_approve: true,
441 approval_mode: ApprovalMode::Auto,
442 translation_enabled: false,
443 allowed_tools: None,
444 dynamic_tools: Vec::new(),
445 hook_executor: None,
446 verbosity: None,
447 provenance: UserInputProvenance::ExternalUser,
448 submission_id: None,
449 }))
450 .await
451 .expect("send model turn");
452
453 let mut saw_tool_result = false;
454 let mut saw_complete = false;
455 let mut rx = handle.rx_event.write().await;
456 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
457 .await
458 .expect("timed out waiting for engine event")
459 {
460 match event {
461 Event::ApprovalRequired { .. } => {
462 panic!("YOLO mode must not prompt for an ordinary background shell command");
463 }
464 Event::ToolCallComplete { name, result, .. } => {
465 if name == "Bash" {
466 saw_tool_result = true;
467 let result = result.expect("shell result");
468 assert!(result.success, "{result:?}");
469 assert!(
470 result.content.contains("Background task started"),
471 "expected a background start, got: {result:?}"
472 );
473 }
474 }
475 Event::TurnComplete { status, .. } => {
476 assert_eq!(status, TurnOutcomeStatus::Completed);
477 saw_complete = true;
478 break;
479 }
480 _ => {}
481 }
482 }
483 drop(rx);
484
485 handle.send(Op::Shutdown).await.expect("shutdown engine");
486 run_task.await.expect("engine task");
487 assert!(saw_tool_result);
488 assert!(saw_complete);
489 }
490
491 #[tokio::test]
492 #[allow(clippy::await_holding_lock)]
493 async fn yolo_mode_executes_publish_like_shell_without_prompt() {
494 // #4595: Full Access (Bypass/YOLO) is truly full access — the publish
495 // floor prompts only in Ask/Auto-Review postures. The regression guard is
496 // the absence of ApprovalRequired; execution itself may fail (the tempdir
497 // is not a git repo), which is fine.
498 use wiremock::matchers::{body_string_contains, method, path};
499 use wiremock::{Mock, MockServer, ResponseTemplate};
500
501 let _lock = lock_test_env();
502 let workspace = tempdir().expect("tempdir");
503 let server = MockServer::start().await;
504
505 let tool_call_sse = concat!(
506 "data: {\"id\":\"chatcmpl-publish\",\"choices\":[{\"index\":0,\"delta\":{\"tool_calls\":[",
507 "{\"index\":0,\"id\":\"call_publish\",\"type\":\"function\",\"function\":{\"name\":\"Bash\",",
508 "\"arguments\":\"{\\\"action\\\":\\\"run\\\",\\\"command\\\":\\\"git push origin main\\\",\\\"background\\\":true}\"}}",
509 "]},\"finish_reason\":null}]}\n\n",
510 "data: {\"id\":\"chatcmpl-publish\",\"choices\":[{\"index\":0,\"delta\":{},",
511 "\"finish_reason\":\"tool_calls\"}]}\n\n",
512 "data: [DONE]\n\n",
513 );
514 let done_sse = concat!(
515 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
516 "\"delta\":{\"content\":\"ack\"},\"finish_reason\":null}]}\n\n",
517 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
518 "\"finish_reason\":\"stop\"}]}\n\n",
519 "data: [DONE]\n\n",
520 );
521
522 Mock::given(method("POST"))
523 .and(path("/v1/chat/completions"))
524 .and(body_string_contains("call_publish"))
525 .respond_with(
526 ResponseTemplate::new(200)
527 .insert_header("content-type", "text/event-stream")
528 .set_body_string(done_sse),
529 )
530 .expect(1)
531 .with_priority(1)
532 .mount(&server)
533 .await;
534 Mock::given(method("POST"))
535 .and(path("/v1/chat/completions"))
536 .respond_with(
537 ResponseTemplate::new(200)
538 .insert_header("content-type", "text/event-stream")
539 .set_body_string(tool_call_sse),
540 )
541 .expect(1)
542 .with_priority(2)
543 .mount(&server)
544 .await;
545
546 let api_config = Config {
547 ..Config::default()
548 }
549 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
550 let (engine, handle) = Engine::new(
551 EngineConfig {
552 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
553 workspace: workspace.path().to_path_buf(),
554 snapshots_enabled: false,
555 subagents_enabled: false,
556 ..EngineConfig::default()
557 },
558 &api_config,
559 );
560 let run_task = tokio::spawn(engine.run());
561
562 handle
563 .send(Op::SendMessage(TurnSpec {
564 max_output_tokens: None,
565 content: "please publish this crate".to_string(),
566 images: Vec::new(),
567 mode: AppMode::Agent,
568 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
569 compaction: Box::new(CompactionConfig::default()),
570 initial_routed_usage: Box::default(),
571 goal_objective: None,
572 goal_token_budget: None,
573 goal_status: crate::tools::goal::GoalStatus::Active,
574 reasoning_effort: None,
575 reasoning_effort_auto: false,
576 auto_model: false,
577 allow_shell: true,
578 trust_mode: true,
579 auto_approve: true,
580 approval_mode: ApprovalMode::Bypass,
581 translation_enabled: false,
582 allowed_tools: None,
583 dynamic_tools: Vec::new(),
584 hook_executor: None,
585 verbosity: None,
586 provenance: UserInputProvenance::ExternalUser,
587 submission_id: None,
588 }))
589 .await
590 .expect("send model turn");
591
592 let mut saw_tool_complete = false;
593 let mut saw_complete = false;
594 let mut rx = handle.rx_event.write().await;
595 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
596 .await
597 .expect("timed out waiting for engine event")
598 {
599 match event {
600 Event::ApprovalRequired {
601 tool_name,
602 description,
603 ..
604 } => {
605 panic!(
606 "Full Access must not prompt for publish-like shell \
607 (#4595); got prompt for {tool_name}: {description}"
608 );
609 }
610 Event::ToolCallComplete { name, .. } if name == "Bash" => {
611 // Execution outcome is irrelevant (the tempdir is not a git
612 // repo); the contract is that it ran without a prompt.
613 saw_tool_complete = true;
614 }
615 Event::TurnComplete { status, .. } => {
616 assert_eq!(status, TurnOutcomeStatus::Completed);
617 saw_complete = true;
618 break;
619 }
620 _ => {}
621 }
622 }
623 drop(rx);
624
625 handle.send(Op::Shutdown).await.expect("shutdown engine");
626 run_task.await.expect("engine task");
627 assert!(
628 saw_tool_complete,
629 "the publish-like shell should execute without a prompt under Full Access"
630 );
631 assert!(saw_complete, "the publish-like turn should complete");
632 }
633
634 #[tokio::test]
635 #[allow(clippy::await_holding_lock)]
636 async fn yolo_mode_does_not_prompt_for_mcp_action() {
637 // #3790: MCP mutations are governed by the selected mode, just like shell.
638 // YOLO must not emit an approval request for a non-read-only MCP tool; this
639 // fixture has no GitHub MCP server, so execution may fail after the no-prompt
640 // planning decision. The regression guard is the absence of ApprovalRequired.
641 use wiremock::matchers::{body_string_contains, method, path};
642 use wiremock::{Mock, MockServer, ResponseTemplate};
643
644 let _lock = lock_test_env();
645 let workspace = tempdir().expect("tempdir");
646 let server = MockServer::start().await;
647
648 let tool_call_sse = concat!(
649 "data: {\"id\":\"chatcmpl-mcp\",\"choices\":[{\"index\":0,\"delta\":{\"tool_calls\":[",
650 "{\"index\":0,\"id\":\"call_mcp\",\"type\":\"function\",\"function\":{\"name\":\"mcp_github_create_pull_request\",",
651 "\"arguments\":\"{\\\"title\\\":\\\"test\\\",\\\"body\\\":\\\"body\\\"}\"}}",
652 "]},\"finish_reason\":null}]}\n\n",
653 "data: {\"id\":\"chatcmpl-mcp\",\"choices\":[{\"index\":0,\"delta\":{},",
654 "\"finish_reason\":\"tool_calls\"}]}\n\n",
655 "data: [DONE]\n\n",
656 );
657 let done_sse = concat!(
658 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
659 "\"delta\":{\"content\":\"ack\"},\"finish_reason\":null}]}\n\n",
660 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
661 "\"finish_reason\":\"stop\"}]}\n\n",
662 "data: [DONE]\n\n",
663 );
664
665 Mock::given(method("POST"))
666 .and(path("/v1/chat/completions"))
667 .and(body_string_contains("MCP tool failed"))
668 .respond_with(
669 ResponseTemplate::new(200)
670 .insert_header("content-type", "text/event-stream")
671 .set_body_string(done_sse),
672 )
673 .expect(1)
674 .with_priority(1)
675 .mount(&server)
676 .await;
677 Mock::given(method("POST"))
678 .and(path("/v1/chat/completions"))
679 .respond_with(
680 ResponseTemplate::new(200)
681 .insert_header("content-type", "text/event-stream")
682 .set_body_string(tool_call_sse),
683 )
684 .expect(1)
685 .with_priority(2)
686 .mount(&server)
687 .await;
688
689 let api_config = Config {
690 ..Config::default()
691 }
692 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
693 let (engine, handle) = Engine::new(
694 EngineConfig {
695 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
696 workspace: workspace.path().to_path_buf(),
697 snapshots_enabled: false,
698 subagents_enabled: false,
699 ..EngineConfig::default()
700 },
701 &api_config,
702 );
703 let run_task = tokio::spawn(engine.run());
704
705 handle
706 .send(Op::SendMessage(TurnSpec {
707 max_output_tokens: None,
708 content: "please open the PR".to_string(),
709 images: Vec::new(),
710 mode: AppMode::Agent,
711 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
712 compaction: Box::new(CompactionConfig::default()),
713 initial_routed_usage: Box::default(),
714 goal_objective: None,
715 goal_token_budget: None,
716 goal_status: crate::tools::goal::GoalStatus::Active,
717 reasoning_effort: None,
718 reasoning_effort_auto: false,
719 auto_model: false,
720 allow_shell: true,
721 trust_mode: true,
722 auto_approve: true,
723 approval_mode: ApprovalMode::Bypass,
724 translation_enabled: false,
725 allowed_tools: None,
726 dynamic_tools: Vec::new(),
727 hook_executor: None,
728 verbosity: None,
729 provenance: UserInputProvenance::ExternalUser,
730 submission_id: None,
731 }))
732 .await
733 .expect("send model turn");
734
735 let mut saw_mcp_result = false;
736 let mut saw_complete = false;
737 let mut rx = handle.rx_event.write().await;
738 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
739 .await
740 .expect("timed out waiting for engine event")
741 {
742 match event {
743 Event::ApprovalRequired { .. } => {
744 panic!("YOLO mode must not prompt for an MCP action");
745 }
746 Event::ToolCallComplete { name, result, .. }
747 if name == "mcp_github_create_pull_request" =>
748 {
749 saw_mcp_result = true;
750 let err = result
751 .expect_err("unconfigured MCP server should fail after no-prompt planning");
752 assert!(
753 err.to_string().contains("MCP tool failed"),
754 "unexpected MCP error: {err:?}"
755 );
756 }
757 Event::TurnComplete { status, .. } => {
758 assert_eq!(status, TurnOutcomeStatus::Completed);
759 saw_complete = true;
760 break;
761 }
762 _ => {}
763 }
764 }
765 drop(rx);
766
767 handle.send(Op::Shutdown).await.expect("shutdown engine");
768 run_task.await.expect("engine task");
769 assert!(
770 saw_mcp_result,
771 "the MCP tool should execute without an approval gate"
772 );
773 assert!(saw_complete, "the YOLO MCP turn should complete");
774 }
775
776 #[tokio::test]
777 async fn run_shell_command_op_preserves_plan_mode_shell_block() {
778 let (mut engine, handle) = Engine::new(EngineConfig::default(), &Config::default());
779
780 engine
781 .handle_run_shell_command(
782 "echo blocked".to_string(),
783 AppMode::Plan,
784 false,
785 false,
786 false,
787 ApprovalMode::Suggest,
788 )
789 .await;
790
791 let mut saw_complete = false;
792 let mut saw_turn_complete = false;
793 let mut rx = handle.rx_event.write().await;
794 while let Some(event) = rx.recv().await {
795 match event {
796 Event::ApprovalRequired { .. } => {
797 panic!("Plan mode shell should be blocked before approval");
798 }
799 Event::ToolCallComplete { name, result, .. } => {
800 saw_complete = true;
801 assert_eq!(name, "Bash");
802 let err = result.expect_err("plan shell should fail");
803 assert!(
804 err.to_string()
805 .contains("Tool 'bash' is unavailable in Plan mode"),
806 "{err}"
807 );
808 }
809 Event::TurnComplete { status, .. } => {
810 saw_turn_complete = true;
811 assert_eq!(status, TurnOutcomeStatus::Failed);
812 break;
813 }
814 _ => {}
815 }
816 }
817
818 assert!(saw_complete);
819 assert!(saw_turn_complete);
820 }
821
822 #[test]
823 fn deferred_tool_preflight_skips_already_active_tools() {
824 let mut tool = api_tool("deferred_tool");
825 tool.defer_loading = Some(true);
826 let catalog = vec![tool];
827 let mut active = HashSet::from(["deferred_tool".to_string()]);
828
829 assert!(
830 preflight_requested_deferred_tool("deferred_tool", &json!({}), &catalog, &mut active,)
831 .is_none(),
832 "already active tools should execute normally"
833 );
834 }
835
836 #[test]
837 fn turn_tool_registry_builder_keeps_plan_primitive_identity() {
838 let (engine, _handle) = Engine::new(EngineConfig::default(), &Config::default());
839 let registry = engine
840 .build_turn_tool_registry_builder(
841 AppMode::Plan,
842 engine.config.todos.clone(),
843 engine.config.plan_state.clone(),
844 )
845 .build(engine.build_tool_context(AppMode::Plan, false));
846
847 for primitive in ["read", "write", "edit", "bash"] {
848 assert!(registry.contains(primitive), "missing {primitive}");
849 }
850 for hidden in ["File", "Bash", "read_file", "write_file", "edit_file"] {
851 assert!(registry.contains(hidden), "missing hidden {hidden}");
852 }
853 assert!(registry.contains("list_dir"));
854 let api_names = registry
855 .to_api_tools()
856 .into_iter()
857 .map(|tool| tool.name)
858 .collect::<HashSet<_>>();
859 for primitive in ["read", "write", "edit", "bash"] {
860 assert!(api_names.contains(primitive), "missing visible {primitive}");
861 }
862 for hidden in ["File", "Bash", "read_file", "write_file", "edit_file"] {
863 assert!(!api_names.contains(hidden), "visible hidden tool {hidden}");
864 }
865 assert!(!registry.contains("exec_shell"));
866 assert!(!registry.contains("exec_shell_wait"));
867 assert!(!registry.contains("exec_shell_interact"));
868 assert!(!registry.contains("task_shell_start"));
869 assert!(!registry.contains("task_create"));
870 assert!(!registry.contains("task_gate_run"));
871 assert!(!registry.contains("rlm"));
872 assert!(!registry.contains("fim_edit"));
873 assert!(registry.contains("update_plan"));
874 assert!(registry.contains("create_goal"));
875 assert!(registry.contains("get_goal"));
876 assert!(registry.contains("update_goal"));
877 assert!(registry.contains("tasks"));
878 assert!(!registry.contains("task_list"));
879 assert!(!registry.contains("task_read"));
880 assert!(registry.contains("handle_read"));
881 assert_eq!(
882 registry.context().shell_policy,
883 crate::worker_profile::ShellPolicy::None
884 );
885 }
886
887 /// Plan mode toggle must not change the byte representation of the tool
888 /// catalog head. DeepSeek's KV prefix cache includes the tools array in
889 /// the immutable prefix; if toggling between Plan and Agent mode changes
890 /// the tool bytes, every mode switch forces a full re-prefill.
891 ///
892 /// This test verifies two invariants:
893 /// 1. Building the catalog twice for the same mode produces identical bytes.
894 /// 2. The head of the catalog (non-deferred tools) preserves its order
895 /// when deferred tools are activated mid-session.
896 #[test]
897 fn plan_mode_toggle_preserves_catalog_byte_stability() {
898 let always_load = HashSet::new();
899
900 // Build catalog for Plan mode twice — must be byte-identical.
901 let plan_native = vec![
902 api_tool("read"),
903 api_tool("write"),
904 api_tool("edit"),
905 api_tool("bash"),
906 api_tool("agent"),
907 api_tool("tool_search"),
908 api_tool("list_dir"),
909 ];
910 let plan_mcp = vec![api_tool("mcp_search"), api_tool("mcp_write")];
911
912 let catalog_a = build_model_tool_catalog(
913 plan_native.clone(),
914 plan_mcp.clone(),
915 AppMode::Plan,
916 &always_load,
917 );
918 let catalog_b = build_model_tool_catalog(
919 plan_native.clone(),
920 plan_mcp.clone(),
921 AppMode::Plan,
922 &always_load,
923 );
924
925 let json_a = serde_json::to_string(&catalog_a).unwrap();
926 let json_b = serde_json::to_string(&catalog_b).unwrap();
927 assert_eq!(
928 json_a, json_b,
929 "building the catalog twice for Plan mode must produce identical bytes"
930 );
931
932 // Build catalog for Agent mode twice — must be byte-identical.
933 let agent_catalog_a = build_model_tool_catalog(
934 plan_native.clone(),
935 plan_mcp.clone(),
936 AppMode::Agent,
937 &always_load,
938 );
939 let agent_catalog_b = build_model_tool_catalog(
940 plan_native.clone(),
941 plan_mcp.clone(),
942 AppMode::Agent,
943 &always_load,
944 );
945
946 let agent_json_a = serde_json::to_string(&agent_catalog_a).unwrap();
947 let agent_json_b = serde_json::to_string(&agent_catalog_b).unwrap();
948 assert_eq!(
949 agent_json_a, agent_json_b,
950 "building the catalog twice for Agent mode must produce identical bytes"
951 );
952
953 // Modes keep the same primitive identities; central authority gates decide
954 // whether an advertised write/edit/bash call may execute.
955 let plan_names: Vec<&str> = catalog_a
956 .iter()
957 .filter(|t| !t.defer_loading.unwrap_or(false))
958 .map(|t| t.name.as_str())
959 .collect();
960 let agent_names: Vec<&str> = agent_catalog_a
961 .iter()
962 .filter(|t| !t.defer_loading.unwrap_or(false))
963 .map(|t| t.name.as_str())
964 .collect();
965
966 let expected_head = ["agent", "bash", "edit", "read", "tool_search", "write"];
967 assert_eq!(plan_names, expected_head);
968 assert_eq!(agent_names, expected_head);
969
970 // Verify that activating a deferred tool mid-session appends to the
971 // tail without reordering the head.
972 let mut tools_with_deferred = plan_native.clone();
973 tools_with_deferred.push({
974 let mut t = api_tool("deferred_search");
975 t.defer_loading = Some(true);
976 t
977 });
978 let catalog_with_deferred = build_model_tool_catalog(
979 tools_with_deferred,
980 plan_mcp.clone(),
981 AppMode::Agent,
982 &always_load,
983 );
984
985 // Activate the deferred tool.
986 let mut active: HashSet<String> = catalog_with_deferred
987 .iter()
988 .filter(|t| !t.defer_loading.unwrap_or(false))
989 .map(|t| t.name.clone())
990 .collect();
991 active.insert("deferred_search".to_string());
992
993 let listed = active_tools_for_step(&catalog_with_deferred, &active);
994 let listed_names: Vec<&str> = listed.iter().map(|t| t.name.as_str()).collect();
995
996 // The head (non-deferred tools) must still be in their original order.
997 let head_names: Vec<&str> = catalog_with_deferred
998 .iter()
999 .filter(|t| !t.defer_loading.unwrap_or(false))
1000 .map(|t| t.name.as_str())
1001 .collect();
1002 assert!(
1003 listed_names.starts_with(&head_names),
1004 "activating a deferred tool must not reorder the catalog head: \
1005 expected {head_names:?} as prefix, got {listed_names:?}"
1006 );
1007 // The deferred tool must be at the tail.
1008 assert_eq!(
1009 listed_names.last(),
1010 Some(&"deferred_search"),
1011 "deferred tool must be appended at the tail"
1012 );
1013 }
1014
1015 #[test]
1016 fn parent_turn_registry_includes_goal_tools_for_all_modes() {
1017 let (engine, _handle) = Engine::new(EngineConfig::default(), &Config::default());
1018
1019 for mode in [AppMode::Plan, AppMode::Agent, AppMode::Operate] {
1020 let registry = engine
1021 .build_turn_tool_registry_builder(
1022 mode,
1023 engine.config.todos.clone(),
1024 engine.config.plan_state.clone(),
1025 )
1026 .build(engine.build_tool_context(mode, false));
1027
1028 for name in ["create_goal", "get_goal", "update_goal"] {
1029 assert!(
1030 registry.contains(name),
1031 "parent {mode:?} registry should expose {name}"
1032 );
1033 }
1034 }
1035 }
1036
1037 #[test]
1038 fn plan_mode_registry_can_expose_agent_launcher_without_shell_tools() {
1039 let tmp = tempdir().expect("tempdir");
1040 let (engine, _handle) = Engine::new(EngineConfig::default(), &Config::default());
1041 let context = engine.build_tool_context(AppMode::Plan, false);
1042 let client = CodewhaleClient::new(
1043 &Config {
1044 ..Config::default()
1045 }
1046 .with_legacy_root(Some("test-key".to_string()), None),
1047 )
1048 .expect("stub client");
1049 let manager = crate::tools::subagent::new_shared_subagent_manager(tmp.path().to_path_buf(), 4);
1050 let mut runtime = SubAgentRuntime::new(
1051 client,
1052 DEFAULT_TEXT_MODEL.to_string(),
1053 context.clone(),
1054 false,
1055 None,
1056 manager.clone(),
1057 )
1058 .with_agent_tool_surface_options(
1059 engine.agent_tool_surface_options(shell_policy_for_mode(AppMode::Plan, false)),
1060 );
1061 runtime.worker_profile = WorkerRuntimeProfile::for_role(FleetRole::Planner);
1062
1063 let registry = engine
1064 .build_turn_tool_registry_builder(
1065 AppMode::Plan,
1066 engine.config.todos.clone(),
1067 engine.config.plan_state.clone(),
1068 )
1069 .with_subagent_tools(manager, runtime)
1070 .build(context);
1071
1072 assert!(
1073 registry.contains("agent"),
1074 "Plan mode should be able to request focused read-only sub-agents"
1075 );
1076 assert!(
1077 !registry.contains("exec_shell"),
1078 "Plan mode must remain shell-free while exposing sub-agent delegation"
1079 );
1080 }
1081
1082 #[test]
1083 fn mode_invariant_matrix_covers_context_catalog_subagents_and_prompt_metadata() {
1084 use crate::sandbox::SandboxPolicy;
1085 use crate::worker_profile::ShellPolicy;
1086 use ApprovalMode;
1087
1088 #[derive(Clone, Copy)]
1089 enum ExpectedSandbox {
1090 ReadOnly,
1091 WorkspaceWrite,
1092 DangerFullAccess,
1093 }
1094
1095 struct ModeCase {
1096 name: &'static str,
1097 mode: AppMode,
1098 shell_policy: ShellPolicy,
1099 sandbox: ExpectedSandbox,
1100 trust_mode: bool,
1101 auto_approve: bool,
1102 approval_mode: ApprovalMode,
1103 plan_hint: bool,
1104 }
1105
1106 let cases = [
1107 ModeCase {
1108 name: "plan",
1109 mode: AppMode::Plan,
1110 shell_policy: ShellPolicy::None,
1111 sandbox: ExpectedSandbox::ReadOnly,
1112 trust_mode: false,
1113 auto_approve: false,
1114 approval_mode: ApprovalMode::Suggest,
1115 plan_hint: true,
1116 },
1117 ModeCase {
1118 name: "agent",
1119 mode: AppMode::Agent,
1120 shell_policy: ShellPolicy::Full,
1121 sandbox: ExpectedSandbox::WorkspaceWrite,
1122 trust_mode: false,
1123 auto_approve: false,
1124 approval_mode: ApprovalMode::Suggest,
1125 plan_hint: false,
1126 },
1127 ModeCase {
1128 name: "agent-full-access",
1129 mode: AppMode::Agent,
1130 shell_policy: ShellPolicy::Full,
1131 sandbox: ExpectedSandbox::DangerFullAccess,
1132 trust_mode: true,
1133 auto_approve: true,
1134 approval_mode: ApprovalMode::Bypass,
1135 plan_hint: false,
1136 },
1137 ModeCase {
1138 name: "operate",
1139 mode: AppMode::Operate,
1140 shell_policy: ShellPolicy::Full,
1141 sandbox: ExpectedSandbox::WorkspaceWrite,
1142 trust_mode: false,
1143 auto_approve: false,
1144 approval_mode: ApprovalMode::Suggest,
1145 plan_hint: false,
1146 },
1147 ];
1148
1149 for case in cases {
1150 let tmp = tempdir().expect("tempdir");
1151 let config = EngineConfig {
1152 workspace: tmp.path().to_path_buf(),
1153 allow_shell: true,
1154 trust_mode: case.trust_mode,
1155 ..EngineConfig::default()
1156 };
1157 let (mut engine, _handle) = Engine::new(config, &Config::default());
1158 engine.current_mode = case.mode;
1159 engine.session.allow_shell = true;
1160 engine.session.trust_mode = case.trust_mode;
1161 engine.session.auto_approve = case.auto_approve;
1162 engine.session.approval_mode = case.approval_mode;
1163
1164 let policy = effective_input_policy(
1165 UserInputProvenance::ExternalUser,
1166 case.mode,
1167 "continue",
1168 engine.session.allow_shell,
1169 engine.session.trust_mode,
1170 engine.session.auto_approve,
1171 engine.session.approval_mode,
1172 );
1173 assert_eq!(policy.mode, case.mode, "{}", case.name);
1174 assert_eq!(policy.trust_mode, case.trust_mode, "{}", case.name);
1175 assert_eq!(policy.auto_approve, case.auto_approve, "{}", case.name);
1176 assert_eq!(policy.approval_mode, case.approval_mode, "{}", case.name);
1177 assert!(policy.allow_shell, "{}", case.name);
1178
1179 let context = engine.build_tool_context(case.mode, case.auto_approve);
1180 assert_eq!(context.shell_policy, case.shell_policy, "{}", case.name);
1181 assert_eq!(context.trust_mode, case.trust_mode, "{}", case.name);
1182 assert_eq!(context.auto_approve, case.auto_approve, "{}", case.name);
1183 assert_eq!(
1184 context.shell_network_denied_hint.is_some(),
1185 case.plan_hint,
1186 "{}",
1187 case.name
1188 );
1189 let sandbox = context
1190 .elevated_sandbox_policy
1191 .as_ref()
1192 .expect("mode context should always carry an elevated sandbox policy");
1193 match (case.sandbox, sandbox) {
1194 (ExpectedSandbox::ReadOnly, SandboxPolicy::ReadOnly) => {}
1195 (
1196 ExpectedSandbox::WorkspaceWrite,
1197 SandboxPolicy::WorkspaceWrite {
1198 writable_roots,
1199 network_access,
1200 ..
1201 },
1202 ) => {
1203 assert_eq!(
1204 writable_roots,
1205 &vec![tmp.path().to_path_buf()],
1206 "{}",
1207 case.name
1208 );
1209 // Workspace-write grants writes, not egress. Network is a
1210 // separate, explicit grant in every mode that reaches here.
1211 assert!(!*network_access, "{}", case.name);
1212 }
1213 (ExpectedSandbox::DangerFullAccess, SandboxPolicy::DangerFullAccess) => {}
1214 _ => panic!("{}: unexpected sandbox policy {sandbox:?}", case.name),
1215 }
1216
1217 let client = CodewhaleClient::new(
1218 &Config {
1219 ..Config::default()
1220 }
1221 .with_legacy_root(Some("test-key".to_string()), None),
1222 )
1223 .expect("stub client");
1224 let manager =
1225 crate::tools::subagent::new_shared_subagent_manager(tmp.path().to_path_buf(), 4);
1226 let mut runtime = SubAgentRuntime::new(
1227 client,
1228 DEFAULT_TEXT_MODEL.to_string(),
1229 context.clone(),
1230 false,
1231 None,
1232 manager.clone(),
1233 )
1234 .with_agent_tool_surface_options(
1235 engine.agent_tool_surface_options(shell_policy_for_mode(case.mode, true)),
1236 );
1237 runtime.worker_profile = WorkerRuntimeProfile::for_role(match case.mode {
1238 AppMode::Plan => FleetRole::Planner,
1239 _ => FleetRole::Worker,
1240 });
1241
1242 let registry = engine
1243 .build_turn_tool_registry_builder(
1244 case.mode,
1245 engine.config.todos.clone(),
1246 engine.config.plan_state.clone(),
1247 )
1248 .with_subagent_tools(manager, runtime)
1249 .build(context);
1250 assert!(registry.contains("agent"), "{}", case.name);
1251 // Primitive identity is mode-stable: both the lowercase `bash` and
1252 // its legacy `Bash` transcript alias register in every mode, and the
1253 // mode gate lives at the execution/catalog boundary (Plan refuses
1254 // shell rather than unregistering the identity).
1255 assert!(registry.contains("bash"), "{}", case.name);
1256 assert!(registry.contains("Bash"), "{}", case.name);
1257 assert!(
1258 !registry.contains("exec_shell"),
1259 "{}: retired exec_shell must remain absent",
1260 case.name
1261 );
1262
1263 let msg = engine.user_text_message_with_turn_metadata_for_route(
1264 "check current policy".to_string(),
1265 DEFAULT_TEXT_MODEL,
1266 false,
1267 None,
1268 false,
1269 );
1270 let metadata = msg.content.last().expect("turn metadata block");
1271 let ContentBlock::Text { text, .. } = metadata else {
1272 panic!("{}: expected text metadata block", case.name);
1273 };
1274 assert!(
1275 text.contains(&format!(
1276 "Current permission posture: {}",
1277 case.approval_mode.permission_chip_label()
1278 )),
1279 "{}: {text}",
1280 case.name
1281 );
1282 // Mode is enforced by runtime policy and the live tool catalog. The
1283 // turn block carries only the independently actionable permission posture.
1284 assert!(
1285 !text.contains("Current mode:"),
1286 "{}: turn metadata must not repeat the mode: {text}",
1287 case.name
1288 );
1289 let prefix = crate::prompts::system_prompt_flat_text(
1290 &crate::prompts::system_prompt_for_mode_with_context_skills_session_and_approval(
1291 &engine.config.workspace,
1292 None,
1293 None,
1294 None,
1295 crate::prompts::PromptSessionContext {
1296 mode: case.mode,
1297 ..Default::default()
1298 },
1299 ),
1300 );
1301 assert!(
1302 !prefix.contains("##### Mode:"),
1303 "{}: mode doctrine must not enter the shared prompt: {prefix}",
1304 case.name
1305 );
1306 }
1307 }
1308
1309 #[test]
1310 fn engine_context_honors_stricter_config_under_full_access() {
1311 use crate::sandbox::SandboxPolicy;
1312 use ApprovalMode;
1313
1314 let tmp = tempdir().expect("tempdir");
1315 let config = EngineConfig {
1316 workspace: tmp.path().to_path_buf(),
1317 ..EngineConfig::default()
1318 };
1319 let api_config = Config {
1320 sandbox_mode: Some("workspace-write".to_string()),
1321 ..Config::default()
1322 };
1323 let (mut engine, _handle) = Engine::new(config, &api_config);
1324 engine.session.approval_mode = ApprovalMode::Bypass;
1325 engine.session.auto_approve = true;
1326
1327 let context = engine.build_tool_context(AppMode::Agent, true);
1328 assert!(matches!(
1329 context.elevated_sandbox_policy.as_ref(),
1330 Some(SandboxPolicy::WorkspaceWrite { writable_roots, .. })
1331 if *writable_roots == vec![tmp.path().to_path_buf()]
1332 ));
1333 }
1334
1335 #[test]
1336 fn mode_invariant_matrix_covers_provenance_authority_narrowing() {
1337 use ApprovalMode;
1338
1339 struct ProvenanceCase {
1340 name: &'static str,
1341 provenance: UserInputProvenance,
1342 expected_mode: AppMode,
1343 expected_trust: bool,
1344 expected_auto: bool,
1345 expected_approval: ApprovalMode,
1346 expect_status: bool,
1347 }
1348
1349 let cases = [
1350 ProvenanceCase {
1351 name: "external user",
1352 provenance: UserInputProvenance::ExternalUser,
1353 expected_mode: AppMode::Agent,
1354 expected_trust: true,
1355 expected_auto: true,
1356 expected_approval: ApprovalMode::Bypass,
1357 expect_status: false,
1358 },
1359 ProvenanceCase {
1360 name: "runtime continuation",
1361 provenance: UserInputProvenance::Runtime,
1362 expected_mode: AppMode::Agent,
1363 expected_trust: true,
1364 expected_auto: true,
1365 expected_approval: ApprovalMode::Bypass,
1366 expect_status: false,
1367 },
1368 ProvenanceCase {
1369 name: "sub-agent handoff",
1370 provenance: UserInputProvenance::SubAgentHandoff,
1371 expected_mode: AppMode::Agent,
1372 expected_trust: true,
1373 expected_auto: true,
1374 expected_approval: ApprovalMode::Bypass,
1375 expect_status: false,
1376 },
1377 ProvenanceCase {
1378 name: "imported transcript",
1379 provenance: UserInputProvenance::ImportedTranscript,
1380 expected_mode: AppMode::Agent,
1381 expected_trust: false,
1382 expected_auto: false,
1383 expected_approval: ApprovalMode::Suggest,
1384 expect_status: true,
1385 },
1386 ProvenanceCase {
1387 name: "memory recall",
1388 provenance: UserInputProvenance::MemoryRecall,
1389 expected_mode: AppMode::Agent,
1390 expected_trust: false,
1391 expected_auto: false,
1392 expected_approval: ApprovalMode::Suggest,
1393 expect_status: true,
1394 },
1395 ProvenanceCase {
1396 name: "assistant generated",
1397 provenance: UserInputProvenance::AssistantGenerated,
1398 expected_mode: AppMode::Agent,
1399 expected_trust: false,
1400 expected_auto: false,
1401 expected_approval: ApprovalMode::Suggest,
1402 expect_status: true,
1403 },
1404 ];
1405
1406 for case in cases {
1407 let policy = effective_input_policy(
1408 case.provenance,
1409 AppMode::Agent,
1410 "continue",
1411 true,
1412 true,
1413 true,
1414 ApprovalMode::Bypass,
1415 );
1416 assert_eq!(policy.mode, case.expected_mode, "{}", case.name);
1417 assert_eq!(policy.trust_mode, case.expected_trust, "{}", case.name);
1418 assert_eq!(policy.auto_approve, case.expected_auto, "{}", case.name);
1419 assert_eq!(
1420 policy.approval_mode, case.expected_approval,
1421 "{}",
1422 case.name
1423 );
1424 assert!(policy.allow_shell, "{}", case.name);
1425 assert_eq!(
1426 policy.status().is_some(),
1427 case.expect_status,
1428 "{}",
1429 case.name
1430 );
1431 }
1432 }
1433
1434 #[test]
1435 fn agent_mode_can_build_auto_approved_tool_context() {
1436 let (engine, _handle) = Engine::new(EngineConfig::default(), &Config::default());
1437
1438 assert!(
1439 !engine
1440 .build_tool_context(AppMode::Agent, false)
1441 .auto_approve
1442 );
1443 assert!(engine.build_tool_context(AppMode::Agent, true).auto_approve);
1444 }
1445
1446 #[test]
1447 fn build_tool_context_preserves_read_snapshots_across_turns() {
1448 let workspace = tempdir().expect("tempdir");
1449 let path = workspace.path().join("observed.txt");
1450 fs::write(&path, "before\n").expect("write fixture");
1451 let config = EngineConfig {
1452 workspace: workspace.path().to_path_buf(),
1453 ..EngineConfig::default()
1454 };
1455 let (engine, _handle) = Engine::new(config, &Config::default());
1456
1457 let read_turn = engine.build_tool_context(AppMode::Agent, false);
1458 read_turn.note_file_read(&path);
1459
1460 let later_turn = engine.build_tool_context(AppMode::Agent, false);
1461 later_turn
1462 .require_fresh_file_read(&path, "observed.txt")
1463 .expect("a later turn should retain the session's fresh read snapshot");
1464
1465 fs::write(&path, "changed contents\n").expect("change fixture");
1466 let err = later_turn
1467 .require_fresh_file_read(&path, "observed.txt")
1468 .expect_err("a retained snapshot must still reject stale edits");
1469 // Names the tool the model can actually call. `read_file` is a retired name
1470 // and pointing a stale-read refusal at it sent the model to a tool that does
1471 // not exist — the guard-then-bad-advice chain this release set out to close.
1472 assert!(
1473 err.to_string()
1474 .contains("changed since the last File action=\"read\" call"),
1475 "stale-read refusal must name a live tool, got: {err}"
1476 );
1477 }
1478
1479 #[test]
1480 fn build_tool_context_uses_typed_shell_policy_per_mode() {
1481 let mut config = EngineConfig {
1482 allow_shell: true,
1483 ..EngineConfig::default()
1484 };
1485 let (engine, _handle) = Engine::new(config.clone(), &Config::default());
1486
1487 // Plan mode is shell-free and exposes no shell tools.
1488 assert_eq!(
1489 engine.build_tool_context(AppMode::Plan, false).shell_policy,
1490 crate::worker_profile::ShellPolicy::None
1491 );
1492 assert_eq!(
1493 engine
1494 .build_tool_context(AppMode::Agent, false)
1495 .shell_policy,
1496 crate::worker_profile::ShellPolicy::Full
1497 );
1498
1499 config.allow_shell = false;
1500 let (engine, _handle) = Engine::new(config, &Config::default());
1501 assert_eq!(
1502 engine
1503 .build_tool_context(AppMode::Agent, false)
1504 .shell_policy,
1505 crate::worker_profile::ShellPolicy::None
1506 );
1507 }
1508
1509 #[test]
1510 fn turn_tool_context_uses_planned_authority_and_route_not_installed_session() {
1511 let (mut engine, _handle) = Engine::new(EngineConfig::default(), &Config::default());
1512 engine.session.allow_shell = false;
1513 engine.session.trust_mode = false;
1514 engine.session.model = "installed-old-model".to_string();
1515
1516 let authority = crate::core::authority::TurnAuthority::from_effective_fields(
1517 AppMode::Agent,
1518 true,
1519 true,
1520 true,
1521 ApprovalMode::Bypass,
1522 );
1523 let route = TurnRouteContext {
1524 provider: ProviderKind::Deepseek,
1525 model: "planned-next-model".to_string(),
1526 capabilities: codewhale_config::route::RouteCapabilities::default(),
1527 limits: Some(codewhale_config::route::RouteLimits {
1528 context_tokens: Some(123_456),
1529 input_tokens: None,
1530 output_tokens: Some(4_096),
1531 }),
1532 client: None,
1533 api_config: Box::new(Config::default()),
1534 locale_tag: engine.config.locale_tag.clone(),
1535 role_models: engine.subagent_role_models(),
1536 auto_model: false,
1537 reasoning_effort: None,
1538 reasoning_effort_auto: false,
1539 };
1540
1541 let context = engine.build_tool_context_for_turn(&authority, &route);
1542 assert_eq!(
1543 context.shell_policy,
1544 crate::worker_profile::ShellPolicy::Full
1545 );
1546 assert!(context.trust_mode);
1547 assert!(context.auto_approve);
1548 assert_eq!(
1549 context.approval_mode,
1550 ApprovalMode::Bypass,
1551 "the turn's posture travels with the context its tools see"
1552 );
1553 // Auto-Review is the case the legacy bit cannot express: folding `false`
1554 // alone would read as Ask, so this is what proves the posture itself is
1555 // carried rather than re-derived.
1556 let auto_review = crate::core::authority::TurnAuthority::from_effective_fields(
1557 AppMode::Agent,
1558 true,
1559 false,
1560 false,
1561 ApprovalMode::Auto,
1562 );
1563 assert_eq!(
1564 engine
1565 .build_tool_context_for_turn(&auto_review, &route)
1566 .approval_mode,
1567 ApprovalMode::Auto
1568 );
1569 assert_eq!(context.route_context_window, Some(123_456));
1570 assert_eq!(context.route_capabilities, route.capabilities);
1571 assert_eq!(
1572 context
1573 .session_objects
1574 .as_ref()
1575 .expect("session object snapshot")
1576 .model,
1577 "planned-next-model"
1578 );
1579 }
1579 lines RUST