返回 CodeWhale
test_cases_11.rs
根目录 / crates / tui / src / core / engine / tests / test_cases_11.rs
1
2
3 #[test]
4 fn model_catalog_exposes_work_update_as_sole_progress_surface() {
5 // #4132: ordinary progress is one model-visible and executable tool (todo_write).
6 let (engine, _handle) = Engine::new(EngineConfig::default(), &Config::default());
7 let registry = engine
8 .build_turn_tool_registry_builder(
9 AppMode::Agent,
10 engine.config.todos.clone(),
11 engine.config.plan_state.clone(),
12 )
13 .build(engine.build_tool_context(AppMode::Agent, false));
14 let always_load = HashSet::new();
15 let catalog = build_model_tool_catalog(
16 registry.to_api_tools_with_cache(true),
17 vec![],
18 AppMode::Agent,
19 &always_load,
20 );
21 let active = initial_active_tools(&catalog);
22 let catalog_names: HashSet<&str> = catalog.iter().map(|tool| tool.name.as_str()).collect();
23
24 assert!(
25 catalog_names.contains("todo_write"),
26 "todo_write must be model-visible"
27 );
28 assert!(
29 active.contains("todo_write"),
30 "todo_write must be available without a discovery turn"
31 );
32 assert!(
33 !catalog_names.contains("update_plan"),
34 "retired Strategy/Plan must stay replay-only"
35 );
36 // Actually registered hidden aliases (work_update family + checklist_write/update)
37 // remain callable via registry but hidden from catalog. Others were never
38 // registered and must stay not callable.
39 for retired in [
40 "work_update",
41 "TodoWrite",
42 "todo",
43 "checklist_write",
44 "checklist_update",
45 ] {
46 assert!(
47 registry.contains(retired),
48 "{retired} hidden alias must remain callable"
49 );
50 assert!(
51 !catalog_names.contains(retired),
52 "{retired} must not appear in the model catalog"
53 );
54 }
55 for retired in [
56 "checklist_add",
57 "checklist_list",
58 "todo_add",
59 "todo_update",
60 "todo_list",
61 ] {
62 assert!(
63 !registry.contains(retired),
64 "{retired} must not be callable"
65 );
66 assert!(
67 !catalog_names.contains(retired),
68 "{retired} must not appear in the model catalog"
69 );
70 }
71 for retired in [
72 "checklist_write",
73 "checklist_add",
74 "checklist_update",
75 "checklist_list",
76 "work_update",
77 "TodoWrite",
78 "todo",
79 "todo_add",
80 "todo_update",
81 "todo_list",
82 ] {
83 assert!(
84 preflight_requested_deferred_tool(
85 retired,
86 &json!({
87 "todos": [
88 { "content": "should not hydrate hidden alias", "status": "completed" }
89 ]
90 }),
91 &catalog,
92 &mut active.clone(),
93 )
94 .is_none(),
95 "{retired} must not have a deferred catalog preflight path"
96 );
97 }
98 }
99
100 #[test]
101 fn user_shell_turn_outcome_distinguishes_cancel_failure_and_success() {
102 let cancelled = Ok(
103 ToolResult::error("Command canceled; process killed.").with_metadata(json!({
104 "status": "Killed",
105 "canceled": true,
106 })),
107 );
108 assert_eq!(
109 user_shell_turn_outcome(&cancelled, false),
110 TurnOutcomeStatus::Interrupted
111 );
112
113 let cancelled_while_awaiting_approval = Err(ToolError::execution_failed(
114 "Request cancelled while awaiting approval",
115 ));
116 assert_eq!(
117 user_shell_turn_outcome(&cancelled_while_awaiting_approval, true),
118 TurnOutcomeStatus::Interrupted
119 );
120
121 let failed = Ok(ToolResult::error("Command failed (exit code: 1)"));
122 assert_eq!(
123 user_shell_turn_outcome(&failed, false),
124 TurnOutcomeStatus::Failed
125 );
126
127 let execution_error = Err(ToolError::execution_failed("shell manager unavailable"));
128 assert_eq!(
129 user_shell_turn_outcome(&execution_error, false),
130 TurnOutcomeStatus::Failed
131 );
132
133 let completed = Ok(ToolResult::success("done"));
134 assert_eq!(
135 user_shell_turn_outcome(&completed, true),
136 TurnOutcomeStatus::Interrupted
137 );
138 assert_eq!(
139 user_shell_turn_outcome(&completed, false),
140 TurnOutcomeStatus::Completed
141 );
142 }
143
144 /// #5191: a user-typed `!` command is pre-approved by provenance — typing it
145 /// IS the approval. It must run without the tool-approval modal even in an
146 /// Ask/Suggest session, and the audit trail must record the user provenance.
147 #[tokio::test]
148 async fn run_shell_command_op_executes_without_approval_modal() {
149 let _guard = lock_test_env();
150 let tmp = tempdir().expect("tempdir");
151 let audit_path = tmp.path().join("tool-audit.jsonl");
152 let _audit = EnvVarGuard::set("CODEWHALE_TOOL_AUDIT_LOG", &audit_path);
153 let (mut engine, handle) = Engine::new(EngineConfig::default(), &Config::default());
154 engine.session.allow_shell = false;
155 engine.config.allow_shell = false;
156
157 engine
158 .handle_run_shell_command(
159 "echo bang-ok".to_string(),
160 AppMode::Agent,
161 true,
162 false,
163 false,
164 ApprovalMode::Suggest,
165 )
166 .await;
167
168 let mut saw_started = false;
169 let mut saw_approval = false;
170 let mut saw_complete = false;
171 let mut saw_turn_complete = false;
172 let mut rx = handle.rx_event.write().await;
173 while let Some(event) = rx.recv().await {
174 match event {
175 Event::TurnStarted { turn_id, route, .. } => {
176 assert!(turn_id.starts_with(USER_SHELL_TOOL_ID_PREFIX));
177 assert!(route.is_none());
178 }
179 Event::ToolCallStarted {
180 id,
181 name,
182 input,
183 model_call,
184 } => {
185 saw_started = true;
186 assert!(model_call.is_none());
187 assert!(id.starts_with(USER_SHELL_TOOL_ID_PREFIX));
188 assert_eq!(name, "Bash");
189 assert_eq!(input["action"], json!("run"));
190 assert_eq!(input["command"], json!("echo bang-ok"));
191 assert_eq!(input["source"], json!("user"));
192 }
193 Event::ApprovalRequired { .. } => {
194 saw_approval = true;
195 }
196 Event::ToolCallComplete {
197 id,
198 name,
199 result,
200 model_call,
201 } => {
202 saw_complete = true;
203 assert!(model_call.is_none());
204 assert!(id.starts_with(USER_SHELL_TOOL_ID_PREFIX));
205 assert_eq!(name, "Bash");
206 let result = result.expect("shell result");
207 assert!(result.success, "{result:?}");
208 assert!(result.content.contains("bang-ok"), "{result:?}");
209 }
210 Event::TurnComplete { status, .. } => {
211 saw_turn_complete = true;
212 assert_eq!(status, TurnOutcomeStatus::Completed);
213 break;
214 }
215 _ => {}
216 }
217 }
218 drop(rx);
219
220 assert!(saw_started);
221 assert!(
222 !saw_approval,
223 "user-typed bang commands must not raise the approval modal (#5191)"
224 );
225 assert!(saw_complete);
226 assert!(saw_turn_complete);
227
228 let audit = std::fs::read_to_string(&audit_path).expect("audit log written");
229 assert!(
230 audit.contains("tool.user_provenance_preapproved"),
231 "audit trail must record the user-provenance pre-approval: {audit}"
232 );
233 assert!(
234 audit.contains("composer_bang"),
235 "audit row must name the composer-bang source: {audit}"
236 );
237 }
238
239 #[tokio::test]
240 async fn run_shell_command_op_skips_approval_when_auto_approved() {
241 let workspace = tempdir().expect("tempdir");
242 let todos = crate::tools::todo::new_shared_todo_list();
243 let plan = crate::tools::plan::new_shared_plan_state();
244 let work = crate::work_graph::new_shared_work_runtime(todos, plan);
245 let runtime_services = crate::tools::spec::RuntimeToolServices {
246 work: Some(work.clone()),
247 ..Default::default()
248 };
249 let (mut engine, handle) = Engine::new(
250 EngineConfig {
251 workspace: workspace.path().to_path_buf(),
252 snapshots_enabled: false,
253 runtime_services,
254 ..EngineConfig::default()
255 },
256 &Config::default(),
257 );
258 let session_id = engine.session.id.clone();
259
260 engine
261 .handle_run_shell_command(
262 "echo bang-yolo".to_string(),
263 AppMode::Agent,
264 true,
265 true,
266 true,
267 ApprovalMode::Auto,
268 )
269 .await;
270
271 let mut saw_complete = false;
272 let mut rx = handle.rx_event.write().await;
273 while let Some(event) = rx.recv().await {
274 match event {
275 Event::ApprovalRequired { .. } => {
276 panic!("auto-approved shell shortcut should not request approval");
277 }
278 Event::ToolCallComplete { result, .. } => {
279 saw_complete = true;
280 let result = result.expect("shell result");
281 assert!(result.success, "{result:?}");
282 assert!(result.content.contains("bang-yolo"), "{result:?}");
283 }
284 Event::TurnComplete { status, .. } => {
285 assert_eq!(status, TurnOutcomeStatus::Completed);
286 break;
287 }
288 _ => {}
289 }
290 }
291
292 assert!(saw_complete);
293 let graph = work
294 .capture(Some(&session_id))
295 .expect("capture bang-shell work")
296 .expect("bang-shell graph")
297 .graph;
298 let operation = graph
299 .nodes
300 .iter()
301 .find(|node| node.kind == crate::work_graph::NodeKind::Operation)
302 .expect("bang-shell operation registered before execution");
303 assert_eq!(operation.state, crate::work_graph::NodeState::Completed);
304 let observation = operation
305 .binding
306 .as_ref()
307 .and_then(|binding| binding.last_observation.as_ref())
308 .expect("terminal shell owner observation");
309 assert!(
310 observation
311 .output
312 .as_ref()
313 .and_then(crate::work_graph::EvidenceRef::raw_bytes)
314 .is_some_and(|raw_bytes| raw_bytes > 0),
315 "bang-shell completion must retain a logical byte-count receipt"
316 );
317 }
318
319 #[tokio::test]
320 async fn run_shell_command_op_allows_readonly_shell_in_auto_mode() {
321 let (mut engine, handle) = Engine::new(EngineConfig::default(), &Config::default());
322 let handle_for_approval = handle.clone();
323
324 let task = tokio::spawn(async move {
325 engine
326 .handle_run_shell_command(
327 "pwd".to_string(),
328 AppMode::Agent,
329 true,
330 false,
331 false,
332 ApprovalMode::Auto,
333 )
334 .await;
335 });
336
337 let mut saw_approval = false;
338 let mut saw_complete = false;
339 let mut rx = handle.rx_event.write().await;
340 while let Some(event) = rx.recv().await {
341 match event {
342 Event::ApprovalRequired { id, .. } => {
343 saw_approval = true;
344 handle_for_approval
345 .approve_tool_call(id)
346 .await
347 .expect("approve unexpected shell prompt");
348 }
349 Event::ToolCallComplete { result, .. } => {
350 saw_complete = true;
351 let result = result.expect("shell result");
352 assert!(result.success, "{result:?}");
353 }
354 Event::TurnComplete { status, .. } => {
355 assert_eq!(status, TurnOutcomeStatus::Completed);
356 break;
357 }
358 _ => {}
359 }
360 }
361 drop(rx);
362 task.await.expect("shell op task");
363
364 assert!(
365 !saw_approval,
366 "read-only shell shortcut should not request approval in Auto mode"
367 );
368 assert!(saw_complete);
369 }
370
371 #[tokio::test]
372 async fn yolo_mode_does_not_prompt_for_typed_ask_rule() {
373 // #3386: a command matching a typed ask-rule (permissions.toml) must not
374 // surface an approval modal in the Full Access posture, even though the
375 // stale ApprovalMode::Auto maps to OnFailure in the execpolicy (honors
376 // ask-rules). The auto_review safety floor and typed deny rules still
377 // apply; only the ask-rule Prompt is suppressed under Full Access.
378 let (mut engine, handle) = Engine::new(
379 EngineConfig {
380 exec_policy_engine: ask_rule_engine("echo"),
381 ..EngineConfig::default()
382 },
383 &Config::default(),
384 );
385
386 engine
387 .handle_run_shell_command(
388 "echo yolo-ask-rule".to_string(),
389 AppMode::Agent,
390 true,
391 true,
392 true,
393 ApprovalMode::Auto,
394 )
395 .await;
396
397 let mut saw_complete = false;
398 let mut rx = handle.rx_event.write().await;
399 while let Some(event) = rx.recv().await {
400 match event {
401 Event::ApprovalRequired { .. } => {
402 panic!("YOLO mode must not prompt for a typed ask-rule");
403 }
404 Event::ToolCallComplete { result, .. } => {
405 saw_complete = true;
406 let result = result.expect("shell result");
407 assert!(result.success, "{result:?}");
408 assert!(result.content.contains("yolo-ask-rule"), "{result:?}");
409 }
410 Event::TurnComplete { status, .. } => {
411 assert_eq!(status, TurnOutcomeStatus::Completed);
412 break;
413 }
414 _ => {}
415 }
416 }
417
418 assert!(saw_complete);
419 }
420
421 #[tokio::test]
422 #[allow(clippy::await_holding_lock)]
423 async fn operate_model_shell_uses_normal_approval_and_workspace_sandbox() {
424 use wiremock::matchers::{body_string_contains, method, path};
425 use wiremock::{Mock, MockServer, ResponseTemplate};
426
427 let _lock = lock_test_env();
428 let workspace = tempdir().expect("tempdir");
429 let server = MockServer::start().await;
430
431 let tool_call_sse = concat!(
432 "data: {\"id\":\"chatcmpl-operate-tools\",\"choices\":[{\"index\":0,\"delta\":{\"tool_calls\":[",
433 "{\"index\":0,\"id\":\"call_operate_shell\",\"type\":\"function\",\"function\":{\"name\":\"Bash\",",
434 "\"arguments\":\"{\\\"action\\\":\\\"run\\\",\\\"command\\\":\\\"echo operate-approved > operate-mode-approved.txt\\\"}\"}}",
435 "]},\"finish_reason\":null}]}\n\n",
436 "data: {\"id\":\"chatcmpl-operate-tools\",\"choices\":[{\"index\":0,\"delta\":{},",
437 "\"finish_reason\":\"tool_calls\"}]}\n\n",
438 "data: [DONE]\n\n",
439 );
440 let done_sse = concat!(
441 "data: {\"id\":\"chatcmpl-operate-done\",\"choices\":[{\"index\":0,",
442 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
443 "data: {\"id\":\"chatcmpl-operate-done\",\"choices\":[{\"index\":0,\"delta\":{},",
444 "\"finish_reason\":\"stop\"}]}\n\n",
445 "data: [DONE]\n\n",
446 );
447 // The goal this fixture seals is seeded directly (prose no longer
448 // creates goals since the #6290 rework); after the approved shell runs,
449 // the model seals it through the same `update_goal` tool a live Operate
450 // turn uses, and only then does the final "done" arrive.
451 let goal_seal_marker = "goal-seal-receipt-0902";
452 let goal_sse = concat!(
453 "data: {\"id\":\"chatcmpl-operate-goal\",\"choices\":[{\"index\":0,\"delta\":{\"tool_calls\":[",
454 "{\"index\":0,\"id\":\"call_operate_goal\",\"type\":\"function\",\"function\":{\"name\":\"update_goal\",",
455 "\"arguments\":\"{\\\"status\\\":\\\"complete\\\",\\\"evidence\\\":\\\"goal-seal-receipt-0902: operate-mode-approved.txt written\\\",",
456 "\\\"verification\\\":{\\\"status\\\":\\\"passed\\\",\\\"check\\\":\\\"cat operate-mode-approved.txt\\\",\\\"summary\\\":\\\"fixture contains operate-approved\\\"}}\"}}",
457 "]},\"finish_reason\":null}]}\n\n",
458 "data: {\"id\":\"chatcmpl-operate-goal\",\"choices\":[{\"index\":0,\"delta\":{},",
459 "\"finish_reason\":\"tool_calls\"}]}\n\n",
460 "data: [DONE]\n\n",
461 );
462
463 Mock::given(method("POST"))
464 .and(path("/v1/chat/completions"))
465 .and(body_string_contains(goal_seal_marker))
466 .and(body_string_contains("tokens_used"))
467 .respond_with(
468 ResponseTemplate::new(200)
469 .insert_header("content-type", "text/event-stream")
470 .set_body_string(done_sse),
471 )
472 .expect(1)
473 .with_priority(1)
474 .mount(&server)
475 .await;
476 // Goal control must execute immediately: continuation cannot depend on
477 // discovering or retrying the very tool that lets the model stop it.
478 Mock::given(method("POST"))
479 .and(path("/v1/chat/completions"))
480 .and(body_string_contains("was deferred and has now been loaded"))
481 .respond_with(
482 ResponseTemplate::new(200)
483 .insert_header("content-type", "text/event-stream")
484 .set_body_string(goal_sse),
485 )
486 .expect(0)
487 .with_priority(2)
488 .mount(&server)
489 .await;
490 Mock::given(method("POST"))
491 .and(path("/v1/chat/completions"))
492 .and(body_string_contains("operate-mode-approved.txt"))
493 .respond_with(
494 ResponseTemplate::new(200)
495 .insert_header("content-type", "text/event-stream")
496 .set_body_string(goal_sse),
497 )
498 .expect(1)
499 .with_priority(3)
500 .mount(&server)
501 .await;
502 Mock::given(method("POST"))
503 .and(path("/v1/chat/completions"))
504 .respond_with(
505 ResponseTemplate::new(200)
506 .insert_header("content-type", "text/event-stream")
507 .set_body_string(tool_call_sse),
508 )
509 .expect(1)
510 .with_priority(3)
511 .mount(&server)
512 .await;
513
514 let api_config = Config {
515 ..Config::default()
516 }
517 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
518 let (engine, handle) = Engine::new(
519 EngineConfig {
520 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
521 workspace: workspace.path().to_path_buf(),
522 snapshots_enabled: false,
523 subagents_enabled: false,
524 terminal_chrome_enabled: false,
525 ..EngineConfig::default()
526 },
527 &api_config,
528 );
529 engine
530 .config
531 .goal_state
532 .lock()
533 .expect("goal lock")
534 .create("write the requested local fixture".to_string(), None)
535 .expect("seed fixture goal");
536 let handle_for_approval = handle.clone();
537 let run_task = tokio::spawn(engine.run());
538
539 handle
540 .send(Op::SendMessage(TurnSpec {
541 max_output_tokens: None,
542 content: "Write the requested local fixture to the workspace".to_string(),
543 images: Vec::new(),
544 mode: AppMode::Operate,
545 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
546 compaction: Box::new(CompactionConfig::default()),
547 initial_routed_usage: Box::default(),
548 goal_objective: Some("write the requested local fixture".to_string()),
549 goal_token_budget: None,
550 goal_status: crate::tools::goal::GoalStatus::Active,
551 reasoning_effort: None,
552 reasoning_effort_auto: false,
553 auto_model: false,
554 allow_shell: true,
555 trust_mode: false,
556 auto_approve: false,
557 approval_mode: ApprovalMode::Suggest,
558 translation_enabled: false,
559 allowed_tools: None,
560 dynamic_tools: Vec::new(),
561 hook_executor: None,
562 verbosity: None,
563 provenance: UserInputProvenance::ExternalUser,
564 submission_id: None,
565 }))
566 .await
567 .expect("send Operate model turn");
568
569 let mut saw_approval = false;
570 let mut saw_shell_result = false;
571 let mut saw_complete = false;
572 let mut rx = handle.rx_event.write().await;
573 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
574 .await
575 .expect("timed out waiting for Operate tool event")
576 {
577 match event {
578 Event::ApprovalRequired { id, tool_name, .. } => {
579 saw_approval = true;
580 assert_eq!(tool_name, "Bash");
581 // The seeded fixture goal is orthogonal to this gate:
582 // Operate uses the normal approval flow either way.
583 handle_for_approval
584 .approve_tool_call(id)
585 .await
586 .expect("approve Operate shell");
587 }
588 Event::ToolCallComplete { name, result, .. } if name == "Bash" => {
589 saw_shell_result = true;
590 let result = result.expect("approved Operate shell result");
591 assert!(result.success, "{result:?}");
592 }
593 Event::TurnComplete { status, .. } => {
594 assert_eq!(status, TurnOutcomeStatus::Completed);
595 saw_complete = true;
596 break;
597 }
598 _ => {}
599 }
600 }
601 drop(rx);
602
603 handle.send(Op::Shutdown).await.expect("shutdown engine");
604 run_task.await.expect("engine task");
605
606 assert!(
607 saw_approval,
608 "Operate should use the normal approval gate instead of a mode-only denial"
609 );
610 assert!(saw_shell_result);
611 assert!(saw_complete);
612 let requests = server.received_requests().await.expect("recorded requests");
613 let first: serde_json::Value = serde_json::from_slice(&requests[0].body).expect("request JSON");
614 for name in ["create_goal", "get_goal", "update_goal"] {
615 assert!(
616 first["tools"]
617 .as_array()
618 .expect("wire tools")
619 .iter()
620 .any(|tool| tool["function"]["name"] == name),
621 "first provider request must expose {name} without discovery"
622 );
623 }
624 let written = std::fs::read_to_string(workspace.path().join("operate-mode-approved.txt"))
625 .expect("workspace-scoped shell output");
626 assert_eq!(written.trim_end(), "operate-approved");
627 }
628
629 /// Drives one model turn whose single `Bash` call needs approval, publishes
630 /// `change_to` (as a runtime PATCH does) while the approval is pending, then
631 /// approves. Returns the call's result and whether the file was written.
632 async fn posture_change_during_approval_wait(
633 change_to: (AppMode, ApprovalMode, bool),
634 ) -> (Result<crate::tools::spec::ToolResult, ToolError>, bool) {
635 use wiremock::matchers::{body_string_contains, method, path};
636 use wiremock::{Mock, MockServer, ResponseTemplate};
637
638 let workspace = tempdir().expect("tempdir");
639 let server = MockServer::start().await;
640 let tool_call_sse = concat!(
641 "data: {\"id\":\"chatcmpl-e2\",\"choices\":[{\"index\":0,\"delta\":{\"tool_calls\":[",
642 "{\"index\":0,\"id\":\"call_e2_shell\",\"type\":\"function\",\"function\":{\"name\":\"Bash\",",
643 "\"arguments\":\"{\\\"action\\\":\\\"run\\\",\\\"command\\\":\\\"echo approved > e2-approved.txt\\\"}\"}}",
644 "]},\"finish_reason\":null}]}\n\n",
645 "data: {\"id\":\"chatcmpl-e2\",\"choices\":[{\"index\":0,\"delta\":{},",
646 "\"finish_reason\":\"tool_calls\"}]}\n\n",
647 "data: [DONE]\n\n",
648 );
649 let done_sse = concat!(
650 "data: {\"id\":\"chatcmpl-e2-done\",\"choices\":[{\"index\":0,",
651 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
652 "data: {\"id\":\"chatcmpl-e2-done\",\"choices\":[{\"index\":0,\"delta\":{},",
653 "\"finish_reason\":\"stop\"}]}\n\n",
654 "data: [DONE]\n\n",
655 );
656 Mock::given(method("POST"))
657 .and(path("/v1/chat/completions"))
658 .and(body_string_contains("call_e2_shell"))
659 .respond_with(
660 ResponseTemplate::new(200)
661 .insert_header("content-type", "text/event-stream")
662 .set_body_string(done_sse),
663 )
664 .with_priority(1)
665 .mount(&server)
666 .await;
667 Mock::given(method("POST"))
668 .and(path("/v1/chat/completions"))
669 .respond_with(
670 ResponseTemplate::new(200)
671 .insert_header("content-type", "text/event-stream")
672 .set_body_string(tool_call_sse),
673 )
674 .expect(1)
675 .with_priority(2)
676 .mount(&server)
677 .await;
678
679 let api_config = Config {
680 ..Config::default()
681 }
682 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
683 let (engine, handle) = Engine::new(
684 EngineConfig {
685 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
686 workspace: workspace.path().to_path_buf(),
687 snapshots_enabled: false,
688 subagents_enabled: false,
689 terminal_chrome_enabled: false,
690 ..EngineConfig::default()
691 },
692 &api_config,
693 );
694 let run_task = tokio::spawn(engine.run());
695 handle
696 .send(Op::SendMessage(TurnSpec {
697 max_output_tokens: None,
698 content: "Record the approval fixture in the workspace".to_string(),
699 images: Vec::new(),
700 mode: AppMode::Agent,
701 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
702 compaction: Box::new(CompactionConfig::default()),
703 initial_routed_usage: Box::default(),
704 goal_objective: None,
705 goal_token_budget: None,
706 goal_status: crate::tools::goal::GoalStatus::Active,
707 reasoning_effort: None,
708 reasoning_effort_auto: false,
709 auto_model: false,
710 allow_shell: true,
711 trust_mode: false,
712 auto_approve: false,
713 approval_mode: ApprovalMode::Suggest,
714 translation_enabled: false,
715 allowed_tools: None,
716 dynamic_tools: Vec::new(),
717 hook_executor: None,
718 verbosity: None,
719 provenance: UserInputProvenance::ExternalUser,
720 submission_id: None,
721 }))
722 .await
723 .expect("send model turn");
724
725 let (mode, approval_mode, auto_approve) = change_to;
726 let mut shell_result = None;
727 let mut rx = handle.rx_event.write().await;
728 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
729 .await
730 .expect("timed out waiting for turn event")
731 {
732 match event {
733 Event::ApprovalRequired { id, .. } => {
734 // The PATCH lands while the approval card is open.
735 handle
736 .try_send(Op::ChangeMode {
737 mode,
738 allow_shell: true,
739 trust_mode: false,
740 auto_approve,
741 approval_mode,
742 configured_sandbox_mode: None,
743 })
744 .expect("publish posture change");
745 handle.approve_tool_call(id).await.expect("approve shell");
746 }
747 Event::ToolCallComplete { name, result, .. } if name == "Bash" => {
748 shell_result = Some(result);
749 }
750 Event::TurnComplete { .. } => break,
751 _ => {}
752 }
753 }
754 drop(rx);
755 handle.send(Op::Shutdown).await.expect("shutdown engine");
756 run_task.await.expect("engine task");
757 let written = workspace.path().join("e2-approved.txt").exists();
758 (shell_result.expect("the approved call completes"), written)
759 }
760
761 #[test]
762 fn live_runtime_authority_narrows_only_when_a_grant_is_withdrawn() {
763 let at = |mode, approval_mode, sandbox: Option<&str>| {
764 LiveRuntimeAuthority::from_fields(
765 mode,
766 true,
767 false,
768 approval_mode == ApprovalMode::Bypass,
769 approval_mode,
770 sandbox.map(str::to_string),
771 )
772 };
773 let ask = at(AppMode::Agent, ApprovalMode::Suggest, None);
774 assert!(!ask.narrows(&ask));
775 assert!(!at(AppMode::Agent, ApprovalMode::Auto, None).narrows(&ask));
776 assert!(!at(AppMode::Agent, ApprovalMode::Bypass, None).narrows(&ask));
777 assert!(!ask.narrows(&at(AppMode::Plan, ApprovalMode::Suggest, None)));
778 assert!(at(AppMode::Plan, ApprovalMode::Suggest, None).narrows(&ask));
779 assert!(at(AppMode::Operate, ApprovalMode::Suggest, None).narrows(&ask));
780 assert!(ask.narrows(&at(AppMode::Agent, ApprovalMode::Bypass, None)));
781 assert!(at(AppMode::Agent, ApprovalMode::Never, None).narrows(&ask));
782 assert!(at(AppMode::Agent, ApprovalMode::Suggest, Some("read-only")).narrows(&ask));
783 assert!(
784 !at(
785 AppMode::Agent,
786 ApprovalMode::Suggest,
787 Some("workspace-write")
788 )
789 .narrows(&at(
790 AppMode::Agent,
791 ApprovalMode::Suggest,
792 Some("read-only")
793 ))
794 );
795 assert!(at(AppMode::Agent, ApprovalMode::Suggest, Some("custom")).narrows(&ask));
796 let mut no_shell = ask.clone();
797 no_shell.allow_shell = false;
798 assert!(no_shell.narrows(&ask));
799 }
800
801 /// E2: approving a call must never invalidate the call it approves. A posture
802 /// PATCH that is equal or broader (Ask -> Auto-Review, Ask -> Full Access)
803 /// while the approval card is open leaves the approved call running.
804 #[tokio::test]
805 #[allow(clippy::await_holding_lock)]
806 async fn broader_posture_patch_during_approval_wait_keeps_the_approved_call() {
807 let _lock = lock_test_env();
808 for change_to in [
809 (AppMode::Agent, ApprovalMode::Auto, false),
810 (AppMode::Agent, ApprovalMode::Bypass, true),
811 (AppMode::Agent, ApprovalMode::Suggest, false),
812 ] {
813 let (result, written) = posture_change_during_approval_wait(change_to).await;
814 let result = result.unwrap_or_else(|err| panic!("{change_to:?}: {err}"));
815 assert!(result.success, "{change_to:?}: {result:?}");
816 assert!(written, "{change_to:?}: the approved shell ran");
817 }
818 }
819
820 /// E2 counterpart: a narrowing PATCH (Work -> Plan, Ask -> Never) still sends
821 /// the approved call back to the model instead of running it under a grant
822 /// the user has since withdrawn.
823 #[tokio::test]
824 #[allow(clippy::await_holding_lock)]
825 async fn narrower_posture_patch_during_approval_wait_fails_the_call() {
826 let _lock = lock_test_env();
827 for change_to in [
828 (AppMode::Plan, ApprovalMode::Suggest, false),
829 (AppMode::Agent, ApprovalMode::Never, false),
830 ] {
831 let (result, written) = posture_change_during_approval_wait(change_to).await;
832 let err = result.expect_err("narrowed posture fails the call");
833 assert!(
834 err.to_string()
835 .contains("Permissions changed before this tool call executed"),
836 "{change_to:?}: {err}"
837 );
838 assert!(!written, "{change_to:?}: the shell must not run");
839 }
840 }
841
842 #[tokio::test]
843 #[allow(clippy::await_holding_lock)]
844 async fn full_access_subagent_handoff_keeps_model_shell_free_of_approval_prompts() {
845 use wiremock::matchers::{body_string_contains, method, path};
846 use wiremock::{Mock, MockServer, ResponseTemplate};
847
848 let _lock = lock_test_env();
849 let workspace = tempdir().expect("tempdir");
850 let server = MockServer::start().await;
851
852 let tool_call_sse = concat!(
853 "data: {\"id\":\"chatcmpl-yolo\",\"choices\":[{\"index\":0,\"delta\":{\"tool_calls\":[",
854 "{\"index\":0,\"id\":\"call_yolo\",\"type\":\"function\",\"function\":{\"name\":\"Bash\",",
855 "\"arguments\":\"{\\\"action\\\":\\\"run\\\",\\\"command\\\":\\\"echo yolo-model-ask-rule\\\"}\"}}",
856 "]},\"finish_reason\":null}]}\n\n",
857 "data: {\"id\":\"chatcmpl-yolo\",\"choices\":[{\"index\":0,\"delta\":{},",
858 "\"finish_reason\":\"tool_calls\"}]}\n\n",
859 "data: [DONE]\n\n",
860 );
861 let done_sse = concat!(
862 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
863 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
864 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
865 "\"finish_reason\":\"stop\"}]}\n\n",
866 "data: [DONE]\n\n",
867 );
868
869 Mock::given(method("POST"))
870 .and(path("/v1/chat/completions"))
871 .and(body_string_contains("yolo-model-ask-rule"))
872 .respond_with(
873 ResponseTemplate::new(200)
874 .insert_header("content-type", "text/event-stream")
875 .set_body_string(done_sse),
876 )
877 .expect(1)
878 .with_priority(1)
879 .mount(&server)
880 .await;
881 Mock::given(method("POST"))
882 .and(path("/v1/chat/completions"))
883 .respond_with(
884 ResponseTemplate::new(200)
885 .insert_header("content-type", "text/event-stream")
886 .set_body_string(tool_call_sse),
887 )
888 .expect(1)
889 .with_priority(2)
890 .mount(&server)
891 .await;
892
893 let api_config = Config {
894 ..Config::default()
895 }
896 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
897 let (engine, handle) = Engine::new(
898 EngineConfig {
899 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
900 workspace: workspace.path().to_path_buf(),
901 snapshots_enabled: false,
902 subagents_enabled: false,
903 exec_policy_engine: ask_rule_engine("echo"),
904 ..EngineConfig::default()
905 },
906 &api_config,
907 );
908 let run_task = tokio::spawn(engine.run());
909
910 handle
911 .send(Op::SendMessage(TurnSpec {
912 max_output_tokens: None,
913 content: "continue from the completed child".to_string(),
914 images: Vec::new(),
915 mode: AppMode::Agent,
916 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
917 compaction: Box::new(CompactionConfig::default()),
918 initial_routed_usage: Box::default(),
919 goal_objective: None,
920 goal_token_budget: None,
921 goal_status: crate::tools::goal::GoalStatus::Active,
922 reasoning_effort: None,
923 reasoning_effort_auto: false,
924 auto_model: false,
925 allow_shell: true,
926 trust_mode: true,
927 // Exercise the valid legacy/host shape where the named posture is
928 // authoritative but the redundant bit is stale.
929 auto_approve: false,
930 approval_mode: ApprovalMode::Bypass,
931 translation_enabled: false,
932 allowed_tools: None,
933 dynamic_tools: Vec::new(),
934 hook_executor: None,
935 verbosity: None,
936 provenance: UserInputProvenance::SubAgentHandoff,
937 submission_id: None,
938 }))
939 .await
940 .expect("send model turn");
941
942 let mut saw_complete = false;
943 let mut rx = handle.rx_event.write().await;
944 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
945 .await
946 .expect("timed out waiting for engine event")
947 {
948 match event {
949 Event::ApprovalRequired { .. } => {
950 panic!("Full Access child handoff must not prompt for an ordinary shell call");
951 }
952 Event::ToolCallComplete { name, result, .. } if name == "Bash" => {
953 saw_complete = true;
954 let result = result.expect("shell result");
955 assert!(result.success, "{result:?}");
956 assert!(result.content.contains("yolo-model-ask-rule"), "{result:?}");
957 }
958 Event::TurnComplete { status, .. } => {
959 assert_eq!(status, TurnOutcomeStatus::Completed);
960 break;
961 }
962 _ => {}
963 }
964 }
965 drop(rx);
966
967 handle.send(Op::Shutdown).await.expect("shutdown engine");
968 run_task.await.expect("engine task");
969 assert!(saw_complete);
970 }
971
972 async fn assert_full_access_model_tool_batch_is_blocked(
973 engine_config: EngineConfig,
974 tool_calls: Vec<(&'static str, serde_json::Value)>,
975 expected_errors: &[(&str, &str)],
976 followup_fragment: &str,
977 ) {
978 use wiremock::matchers::{body_string_contains, method, path};
979 use wiremock::{Mock, MockServer, ResponseTemplate};
980
981 let server = MockServer::start().await;
982 let model_tool_calls = tool_calls
983 .iter()
984 .enumerate()
985 .map(|(index, (name, arguments))| {
986 json!({
987 "index": index,
988 "id": format!("call_full_access_{index}"),
989 "type": "function",
990 "function": {
991 "name": name,
992 "arguments": arguments.to_string(),
993 },
994 })
995 })
996 .collect::<Vec<_>>();
997 let tool_delta = json!({
998 "id": "chatcmpl-full-access-blocked",
999 "choices": [{
1000 "index": 0,
1001 "delta": {"tool_calls": model_tool_calls},
1002 "finish_reason": serde_json::Value::Null,
1003 }],
1004 });
1005 let tool_finish = json!({
1006 "id": "chatcmpl-full-access-blocked",
1007 "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
1008 });
1009 let tool_call_sse = format!("data: {tool_delta}\n\ndata: {tool_finish}\n\ndata: [DONE]\n\n");
1010 let done_sse = concat!(
1011 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
1012 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
1013 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
1014 "\"finish_reason\":\"stop\"}]}\n\n",
1015 "data: [DONE]\n\n",
1016 );
1017
1018 Mock::given(method("POST"))
1019 .and(path("/v1/chat/completions"))
1020 .and(body_string_contains(followup_fragment))
1021 .respond_with(
1022 ResponseTemplate::new(200)
1023 .insert_header("content-type", "text/event-stream")
1024 .set_body_string(done_sse),
1025 )
1026 .expect(1)
1027 .with_priority(1)
1028 .mount(&server)
1029 .await;
1030 Mock::given(method("POST"))
1031 .and(path("/v1/chat/completions"))
1032 .respond_with(
1033 ResponseTemplate::new(200)
1034 .insert_header("content-type", "text/event-stream")
1035 .set_body_string(tool_call_sse),
1036 )
1037 .expect(1)
1038 .with_priority(2)
1039 .mount(&server)
1040 .await;
1041
1042 let api_config = Config {
1043 ..Config::default()
1044 }
1045 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
1046 let (engine, handle) = Engine::new(engine_config, &api_config);
1047 let run_task = tokio::spawn(engine.run());
1048
1049 handle
1050 .send(Op::SendMessage(TurnSpec {
1051 max_output_tokens: None,
1052 content: "exercise the Full Access execution boundary".to_string(),
1053 images: Vec::new(),
1054 mode: AppMode::Agent,
1055 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
1056 compaction: Box::new(CompactionConfig::default()),
1057 initial_routed_usage: Box::default(),
1058 goal_objective: None,
1059 goal_token_budget: None,
1060 goal_status: crate::tools::goal::GoalStatus::Active,
1061 reasoning_effort: None,
1062 reasoning_effort_auto: false,
1063 auto_model: false,
1064 allow_shell: true,
1065 trust_mode: true,
1066 auto_approve: true,
1067 approval_mode: ApprovalMode::Bypass,
1068 translation_enabled: false,
1069 allowed_tools: None,
1070 dynamic_tools: Vec::new(),
1071 hook_executor: None,
1072 verbosity: None,
1073 provenance: UserInputProvenance::ExternalUser,
1074 submission_id: None,
1075 }))
1076 .await
1077 .expect("send Full Access model turn");
1078
1079 let expected = expected_errors.iter().copied().collect::<HashMap<_, _>>();
1080 let mut seen = HashSet::new();
1081 let mut saw_turn_complete = false;
1082 let mut rx = handle.rx_event.write().await;
1083 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
1084 .await
1085 .expect("timed out waiting for Full Access boundary event")
1086 {
1087 match event {
1088 Event::ApprovalRequired { tool_name, .. } => {
1089 panic!("Full Access must not open an approval modal for blocked tool {tool_name}")
1090 }
1091 Event::ToolCallComplete { name, result, .. }
1092 if expected.contains_key(name.as_str()) =>
1093 {
1094 let error = result.expect_err("blocked tool must return an error");
1095 let fragment = expected[name.as_str()];
1096 assert!(
1097 error.to_string().contains(fragment),
1098 "unexpected {name} denial: {error:?}"
1099 );
1100 seen.insert(name);
1101 }
1102 Event::TurnComplete { status, .. } => {
1103 assert_eq!(status, TurnOutcomeStatus::Completed);
1104 saw_turn_complete = true;
1105 break;
1106 }
1107 _ => {}
1108 }
1109 }
1110 drop(rx);
1111
1112 handle.send(Op::Shutdown).await.expect("shutdown engine");
1113 run_task.await.expect("engine task");
1114 assert_eq!(seen.len(), expected.len(), "missing blocked tool results");
1115 assert!(saw_turn_complete);
1116 }
1117
1118 async fn assert_full_access_model_tool_batch_runs(
1119 engine_config: EngineConfig,
1120 tool_calls: Vec<(&'static str, serde_json::Value)>,
1121 expected_names: &[&str],
1122 ) {
1123 use wiremock::matchers::{body_string_contains, method, path};
1124 use wiremock::{Mock, MockServer, ResponseTemplate};
1125
1126 let server = MockServer::start().await;
1127 let model_tool_calls = tool_calls
1128 .iter()
1129 .enumerate()
1130 .map(|(index, (name, arguments))| {
1131 json!({
1132 "index": index,
1133 "id": format!("call_full_access_{index}"),
1134 "type": "function",
1135 "function": {
1136 "name": name,
1137 "arguments": arguments.to_string(),
1138 },
1139 })
1140 })
1141 .collect::<Vec<_>>();
1142 let tool_delta = json!({
1143 "id": "chatcmpl-full-access-blocked",
1144 "choices": [{
1145 "index": 0,
1146 "delta": {"tool_calls": model_tool_calls},
1147 "finish_reason": serde_json::Value::Null,
1148 }],
1149 });
1150 let tool_finish = json!({
1151 "id": "chatcmpl-full-access-blocked",
1152 "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
1153 });
1154 let tool_call_sse = format!("data: {tool_delta}\n\ndata: {tool_finish}\n\ndata: [DONE]\n\n");
1155 // Request 1's "model" response: discover the deferred specialized tools
1156 // through tool_search, exactly as the lowercase contract expects before
1157 // the first direct call.
1158 let search_tool_calls = vec![
1159 json!({
1160 "index": 0,
1161 "id": "call_search_mcp",
1162 "type": "function",
1163 "function": {
1164 "name": "tool_search",
1165 "arguments": r#"{"query":"mcp server"}"#,
1166 },
1167 }),
1168 json!({
1169 "index": 1,
1170 "id": "call_search_rlm",
1171 "type": "function",
1172 "function": {
1173 "name": "tool_search",
1174 "arguments": r#"{"query":"rlm"}"#,
1175 },
1176 }),
1177 ];
1178 let search_delta = json!({
1179 "id": "chatcmpl-full-access-search",
1180 "choices": [{
1181 "index": 0,
1182 "delta": {"tool_calls": search_tool_calls},
1183 "finish_reason": serde_json::Value::Null,
1184 }],
1185 });
1186 let search_finish = json!({
1187 "id": "chatcmpl-full-access-search",
1188 "choices": [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}],
1189 });
1190 let search_sse = format!("data: {search_delta}\n\ndata: {search_finish}\n\ndata: [DONE]\n\n");
1191 let done_sse = concat!(
1192 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,",
1193 "\"delta\":{\"content\":\"done\"},\"finish_reason\":null}]}\n\n",
1194 "data: {\"id\":\"chatcmpl-done\",\"choices\":[{\"index\":0,\"delta\":{},",
1195 "\"finish_reason\":\"stop\"}]}\n\n",
1196 "data: [DONE]\n\n",
1197 );
1198
1199 // wiremock keeps every mounted mock matching even after its expected
1200 // count is reached, and request history accumulates across the three
1201 // steps, so substring matchers on the *calls* would re-fire forever.
1202 // Anchor each mock on the tool-result ids that exist in exactly one
1203 // request body: request 3 carries the executed batch's results, request 2
1204 // carries the search results, and request 1 carries neither.
1205
1206 // Request 3 (exec batch results) terminates with the done turn.
1207 Mock::given(method("POST"))
1208 .and(path("/v1/chat/completions"))
1209 .and(body_string_contains(
1210 "\"tool_call_id\":\"call_full_access_0\"",
1211 ))
1212 .respond_with(
1213 ResponseTemplate::new(200)
1214 .insert_header("content-type", "text/event-stream")
1215 .set_body_string(done_sse),
1216 )
1217 .expect(1)
1218 .with_priority(1)
1219 .mount(&server)
1220 .await;
1221 // Request 2 (search results) executes the deferred specialized tools that
1222 // request 1 discovered through tool_search. The search call activates
1223 // them in the session cache, so this request runs them under Full Access
1224 // without an approval modal.
1225 Mock::given(method("POST"))
1226 .and(path("/v1/chat/completions"))
1227 .and(body_string_contains("\"tool_call_id\":\"call_search_mcp\""))
1228 .respond_with(
1229 ResponseTemplate::new(200)
1230 .insert_header("content-type", "text/event-stream")
1231 .set_body_string(tool_call_sse),
1232 )
1233 .expect(1)
1234 .with_priority(2)
1235 .mount(&server)
1236 .await;
1237 // Request 1: the model discovers the deferred specialized tools it needs.
1238 Mock::given(method("POST"))
1239 .and(path("/v1/chat/completions"))
1240 .respond_with(
1241 ResponseTemplate::new(200)
1242 .insert_header("content-type", "text/event-stream")
1243 .set_body_string(search_sse),
1244 )
1245 .expect(1)
1246 .with_priority(3)
1247 .mount(&server)
1248 .await;
1249
1250 let api_config = Config {
1251 ..Config::default()
1252 }
1253 .with_legacy_root(Some("test-key".to_string()), Some(server.uri()));
1254 let (engine, handle) = Engine::new(engine_config, &api_config);
1255 let run_task = tokio::spawn(engine.run());
1256
1257 handle
1258 .send(Op::SendMessage(TurnSpec {
1259 max_output_tokens: None,
1260 content: "exercise the Full Access auto-approval boundary".to_string(),
1261 images: Vec::new(),
1262 mode: AppMode::Agent,
1263 route: resolved_route_for_test(&api_config, crate::config::DEFAULT_TEXT_MODEL),
1264 compaction: Box::new(CompactionConfig::default()),
1265 initial_routed_usage: Box::default(),
1266 goal_objective: None,
1267 goal_token_budget: None,
1268 goal_status: crate::tools::goal::GoalStatus::Active,
1269 reasoning_effort: None,
1270 reasoning_effort_auto: false,
1271 auto_model: false,
1272 allow_shell: true,
1273 trust_mode: true,
1274 auto_approve: true,
1275 approval_mode: ApprovalMode::Bypass,
1276 translation_enabled: false,
1277 allowed_tools: None,
1278 dynamic_tools: Vec::new(),
1279 hook_executor: None,
1280 verbosity: None,
1281 provenance: UserInputProvenance::ExternalUser,
1282 submission_id: None,
1283 }))
1284 .await
1285 .expect("send Full Access model turn");
1286
1287 let expected = expected_names
1288 .iter()
1289 .copied()
1290 .collect::<std::collections::HashSet<_>>();
1291 let mut seen = HashSet::new();
1292 let mut saw_turn_complete = false;
1293 let mut rx = handle.rx_event.write().await;
1294 while let Some(event) = tokio::time::timeout(model_turn_event_timeout(), rx.recv())
1295 .await
1296 .expect("timed out waiting for Full Access auto-approval event")
1297 {
1298 match event {
1299 Event::ApprovalRequired { tool_name, .. } => {
1300 panic!(
1301 "Full Access must not open an approval modal for auto-approved tool {tool_name}"
1302 )
1303 }
1304 Event::ToolCallComplete { name, result, .. } if expected.contains(name.as_str()) => {
1305 if let Err(error) = &result {
1306 let message = error.to_string();
1307 assert!(
1308 !message.contains("blocked in Full Access"),
1309 "Full Access auto-approves non-bypassable tools: {message}"
1310 );
1311 }
1312 seen.insert(name);
1313 }
1314 Event::TurnComplete { status, error, .. } => {
1315 assert_eq!(
1316 status,
1317 TurnOutcomeStatus::Completed,
1318 "Full Access turn must complete: {error:?}"
1319 );
1320 saw_turn_complete = true;
1321 break;
1322 }
1323 _ => {}
1324 }
1325 }
1326 drop(rx);
1327
1328 handle.send(Op::Shutdown).await.expect("shutdown engine");
1329 run_task.await.expect("engine task");
1330 assert_eq!(
1331 seen.len(),
1332 expected.len(),
1333 "every tool must reach execution, seen: {seen:?}"
1334 );
1335 assert!(saw_turn_complete);
1336 }
1337
1338 #[tokio::test]
1339 #[allow(clippy::await_holding_lock)]
1340 async fn full_access_auto_approves_non_bypassable_registered_tools() {
1341 let _lock = lock_test_env();
1342 let workspace = tempdir().expect("tempdir");
1343 let marker = workspace.path().join("runtime-tool-must-run");
1344 let marker_literal = marker
1345 .to_string_lossy()
1346 .replace('\\', "/")
1347 .replace('\'', "\\'");
1348 // GitHub's Windows image exposes the interpreter as `python`; Unix
1349 // images expose `python3`. Keep the runtime-tool execution receipt
1350 // platform-neutral so this test measures the Full Access boundary rather
1351 // than an executable-name convention.
1352 let python = if cfg!(windows) { "python" } else { "python3" };
1353 let start_probe =
1354 format!("{python} -c \"__import__('pathlib').Path('{marker_literal}').write_text('ran')\"");
1355 let rlm_probe = format!("__import__('pathlib').Path('{marker_literal}').write_text('ran')");
1356 let engine_config = EngineConfig {
1357 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
1358 workspace: workspace.path().to_path_buf(),
1359 mcp_config_path: workspace.path().join("mcp.json"),
1360 snapshots_enabled: false,
1361 subagents_enabled: false,
1362 ..EngineConfig::default()
1363 };
1364 assert_full_access_model_tool_batch_runs(
1365 engine_config,
1366 vec![
1367 (
1368 "start_mcp_server",
1369 json!({"server": start_probe, "name": "auto-approved"}),
1370 ),
1371 (
1372 "rlm",
1373 json!({"action": "eval", "name": "missing-context", "code": rlm_probe}),
1374 ),
1375 ],
1376 &["start_mcp_server", "rlm"],
1377 )
1378 .await;
1379
1380 assert!(
1381 marker.exists(),
1382 "Full Access auto-approves start_mcp_server, so its server command must actually run"
1383 );
1384 }
1385
1386 #[tokio::test]
1387 #[allow(clippy::await_holding_lock)]
1388 async fn full_access_permission_allow_cannot_bypass_repo_law() {
1389 let _lock = lock_test_env();
1390 let workspace = tempdir().expect("tempdir");
1391 let law_dir = workspace.path().join(".codewhale");
1392 fs::create_dir_all(&law_dir).expect("create law directory");
1393 fs::write(
1394 law_dir.join("constitution.json"),
1395 r#"{
1396 "protected_invariants": [{
1397 "text": "Release notes need human review",
1398 "paths": ["CHANGELOG.md"]
1399 }]
1400 }"#,
1401 )
1402 .expect("write repo law fixture");
1403 let target = workspace.path().join("CHANGELOG.md");
1404 let allow_rule = codewhale_execpolicy::ToolAskRule::file_path("write_file", "CHANGELOG.md")
1405 .into_exact_workspace_allow(workspace.path().to_string_lossy().into_owned());
1406 let engine_config = EngineConfig {
1407 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
1408 workspace: workspace.path().to_path_buf(),
1409 snapshots_enabled: false,
1410 subagents_enabled: false,
1411 exec_policy_engine: codewhale_execpolicy::ExecPolicyEngine::with_rulesets(vec![
1412 codewhale_execpolicy::Ruleset::user(vec![], vec![]).with_ask_rules(vec![allow_rule]),
1413 ]),
1414 ..EngineConfig::default()
1415 };
1416 let tool_input =
1417 json!({"action": "write", "filePath": "CHANGELOG.md", "content": "must not be written\n"});
1418 assert_eq!(
1419 file_tool_ask_rule_decision(
1420 &engine_config,
1421 "write_file",
1422 &tool_input,
1423 workspace.path(),
1424 ApprovalMode::Bypass,
1425 ),
1426 Some(ToolAskRuleDecision::Allow),
1427 "precondition: the remembered grant must match before repo law tightens the plan"
1428 );
1429
1430 assert_full_access_model_tool_batch_is_blocked(
1431 engine_config,
1432 vec![("File", tool_input)],
1433 &[(
1434 "File",
1435 "Repository law blocked tool 'File' in Full Access: Repo law holds this write: \"Release notes need human review\"",
1436 )],
1437 "Repository law blocked tool 'File' in Full Access: Repo law holds this write:",
1438 )
1439 .await;
1440
1441 assert!(!target.exists(), "repo-law block must prevent the write");
1442 }
1443
1444 #[tokio::test]
1445 #[allow(clippy::await_holding_lock)]
1446 async fn full_access_file_rules_block_alias_and_parent_paths_without_mutation() {
1447 use codewhale_execpolicy::{ExecPolicyEngine, PermissionAction, Ruleset, ToolAskRule};
1448
1449 let _lock = lock_test_env();
1450 let workspace = tempdir().expect("tempdir");
1451 fs::create_dir(workspace.path().join("sub")).expect("fixture directory");
1452 let target = workspace.path().join("protected.txt");
1453 fs::write(&target, "original\n").expect("fixture contents");
1454 let rules = ["write_file", "edit_file", "apply_patch"]
1455 .into_iter()
1456 .map(|tool| {
1457 let mut rule = ToolAskRule::file_path(tool, "protected.txt");
1458 rule.action = PermissionAction::Deny;
1459 rule
1460 })
1461 .collect();
1462 let engine_config = EngineConfig {
1463 model: crate::config::DEFAULT_TEXT_MODEL.to_string(),
1464 workspace: workspace.path().to_path_buf(),
1465 snapshots_enabled: false,
1466 subagents_enabled: false,
1467 exec_policy_engine: ExecPolicyEngine::with_rulesets(vec![
1468 Ruleset::user(vec![], vec![]).with_ask_rules(rules),
1469 ]),
1470 ..EngineConfig::default()
1471 };
1472 assert_full_access_model_tool_batch_is_blocked(
1473 engine_config,
1474 vec![
1475 ("write_file", json!({"filePath": "protected.txt", "content": "changed\n"})),
1476 ("File", json!({"action": "edit", "file_path": "protected.txt", "search": "original", "replace": "changed"})),
1477 ("apply_patch", json!({"filePath": "protected.txt", "patch": "@@ -1 +1 @@\n-original\n+changed\n"})),
1478 ("write", json!({"path": "sub/../protected.txt", "content": "changed\n"})),
1479 ],
1480 &[
1481 ("write_file", "Permission rule 'tool=write_file path=protected.txt' explicitly denies"),
1482 ("File", "Permission rule 'tool=edit_file path=protected.txt' explicitly denies"),
1483 ("apply_patch", "Permission rule 'tool=apply_patch path=protected.txt' explicitly denies"),
1484 ("write", "Permission rule 'tool=write_file path=protected.txt' explicitly denies"),
1485 ],
1486 "explicitly denies this invocation",
1487 ).await;
1488 assert_eq!(
1489 fs::read_to_string(&target).expect("retained contents"),
1490 "original\n"
1491 );
1492 }
1492 lines RUST