返回 CodeWhale
tests.rs
根目录 / crates / tui / src / tools / shell / tests.rs
1 use super::*;
2
3 use crate::tools::spec::ToolContext;
4 use serde_json::{Value, json};
5 use tempfile::tempdir;
6
7 #[cfg(windows)]
8 use windows::Win32::Foundation::{DUPLICATE_HANDLE_OPTIONS, DuplicateHandle, HANDLE};
9 #[cfg(windows)]
10 use windows::Win32::System::Threading::GetCurrentProcess;
11
12 // `env_lock` serializes tests that mutate the process environment.
13 #[cfg(any(unix, windows))]
14 use std::sync::{Mutex, OnceLock};
15
16 #[cfg(any(unix, windows))]
17 fn env_lock() -> &'static Mutex<()> {
18 static LOCK: OnceLock<Mutex<()>> = OnceLock::new();
19 LOCK.get_or_init(|| Mutex::new(()))
20 }
21
22 const BACKGROUND_COMPLETION_WAIT_MS: u64 = 30_000;
23
24 #[test]
25 fn shell_catalog_guidance_matches_execution() {
26 let tool = LowercaseBashTool;
27 let schema = tool.input_schema();
28 let command = schema["properties"]["command"]["description"]
29 .as_str()
30 .unwrap();
31 let dispatcher = crate::shell_dispatcher::global_dispatcher();
32 assert!(command.contains(dispatcher.kind().binary()));
33 assert!(
34 !tool.description().contains(command),
35 "the command's interpreter guidance must appear once in each request"
36 );
37 assert_eq!(tool.name(), "bash");
38 assert!(tool.model_visible());
39 assert!(!command.contains("action=run"));
40 assert!(!tool.description().contains("background=true"));
41 assert!(
42 tool.description().contains(
43 "In Ask, after a sandbox denial, retry the exact command once with sandbox_permissions (the narrowest wider mode that suffices) and a one-sentence justification; the approval prompt asks the user."
44 ),
45 "foreground guidance must preserve the sandbox retry and approval contract"
46 );
47 let legacy = BashTool::new("Bash");
48 assert!(!legacy.model_visible());
49 assert!(legacy.description().contains(command));
50 assert!(legacy.description().contains("background=true"));
51 assert!(legacy.description().contains("wait=false"));
52 let readonly = BashTool::read_only("Bash");
53 assert!(readonly.description().contains("never through a shell"));
54 assert!(
55 !readonly
56 .input_schema()
57 .to_string()
58 .contains("Actual execution shell")
59 );
60 let alias = BashTool::alias("exec_shell", "run");
61 assert_eq!(alias.description(), legacy.description());
62 let workspace = tempdir().unwrap();
63 let mut registry = crate::tools::ToolRegistry::new(ToolContext::new(workspace.path()));
64 registry.register(std::sync::Arc::new(BashTool::new("Bash")));
65 registry.register(std::sync::Arc::new(LowercaseBashTool));
66 let catalog = registry.to_api_tools();
67 assert_eq!(catalog.len(), 1);
68 assert_eq!(catalog[0].name, "bash");
69 assert_eq!(catalog[0].description, tool.description());
70 assert_eq!(
71 catalog[0].input_schema["properties"]["command"]["description"],
72 command
73 );
74 }
75
76 #[test]
77 #[ignore = "Exports model-visible shell fixtures for opt-in live model evaluation"]
78 fn export_shell_guidance_eval_fixture() {
79 let path = std::env::var_os("SHELL_GUIDANCE_FIXTURE").expect("SHELL_GUIDANCE_FIXTURE");
80 let workspace = tempdir().unwrap();
81 let mut registry = crate::tools::ToolRegistry::new(ToolContext::new(workspace.path()));
82 registry.register(std::sync::Arc::new(BashTool::new("Bash")));
83 registry.register(std::sync::Arc::new(LowercaseBashTool));
84 let catalog = registry.to_api_tools();
85 let tool = catalog.iter().find(|tool| tool.name == "bash").unwrap();
86 let fixture = json!({
87 "name": tool.name,
88 "description": tool.description,
89 "input_schema": tool.input_schema,
90 "shell": crate::shell_dispatcher::global_dispatcher().kind().binary(),
91 });
92 std::fs::write(path, serde_json::to_vec_pretty(&fixture).unwrap()).unwrap();
93 }
94
95 #[test]
96 fn lowercase_bash_schema_is_small_contract() {
97 let schema = LowercaseBashTool.input_schema();
98 assert_eq!(schema["required"], json!(["command"]));
99 assert_eq!(schema["additionalProperties"], false);
100 assert_eq!(
101 schema["properties"]
102 .as_object()
103 .expect("properties")
104 .keys()
105 .cloned()
106 .collect::<std::collections::BTreeSet<_>>(),
107 [
108 "command",
109 "justification",
110 "read_only",
111 "sandbox_permissions",
112 "timeout"
113 ]
114 .into_iter()
115 .map(str::to_string)
116 .collect()
117 );
118 assert!(!BashTool::new("Bash").model_visible());
119 }
120
121 #[test]
122 fn lowercase_bash_description_matches_the_timeout_it_actually_applies() {
123 use super::{
124 CONTRACT_BASH_FOREGROUND_DEFAULT_TIMEOUT_MS, contract_bash_legacy_input,
125 contract_bash_timeout_ms,
126 };
127
128 // `bash {command}` with no `timeout` translates to a legacy input carrying
129 // no `timeout_ms`, and the contract delegate then bounds the foreground run
130 // at the 120 s default and kills the process there.
131 let translated =
132 contract_bash_legacy_input(&json!({"command": "sleep 600"})).expect("translated input");
133 assert!(
134 translated.get("timeout_ms").is_none(),
135 "omitting `timeout` must not synthesise one during translation: {translated}"
136 );
137 assert_eq!(
138 contract_bash_timeout_ms(true, None, false, false),
139 Some(CONTRACT_BASH_FOREGROUND_DEFAULT_TIMEOUT_MS)
140 );
141
142 // The tool description is the only place the model learns this. It used to
143 // say "when omitted there is no default timeout", so a model running a
144 // four-minute build had every reason not to pass a timeout, and got the
145 // process killed at two minutes anyway.
146 let description = LowercaseBashTool.description();
147 assert!(
148 !description.contains("no default timeout"),
149 "description contradicts the applied default: {description}"
150 );
151 let default_seconds = CONTRACT_BASH_FOREGROUND_DEFAULT_TIMEOUT_MS / 1_000;
152 assert!(
153 description.contains(&format!("{default_seconds} seconds")),
154 "description must name the default it applies: {description}"
155 );
156 let schema = LowercaseBashTool.input_schema();
157 let timeout_doc = schema["properties"]["timeout"]["description"]
158 .as_str()
159 .expect("timeout description");
160 assert!(
161 !timeout_doc.contains("no default timeout"),
162 "schema contradicts the applied default: {timeout_doc}"
163 );
164 assert!(
165 timeout_doc.contains(&format!("{default_seconds} seconds")),
166 "schema must name the default it applies: {timeout_doc}"
167 );
168 }
169
170 #[test]
171 fn contract_bash_foreground_without_a_timeout_is_bounded_not_endless() {
172 use super::{
173 BASH_MAX_TIMEOUT_MS, CONTRACT_BASH_FOREGROUND_DEFAULT_TIMEOUT_MS, contract_bash_timeout_ms,
174 };
175
176 // The reported hang: `bash` in the foreground with no timeout. It used to
177 // resolve to BASH_MAX_TIMEOUT_MS (~24.8 days), so an unauthenticated CLI
178 // waiting on a prompt held the turn open indefinitely. It now takes the
179 // default the tool's own schema advertises, which is what arms the
180 // kill-and-rerun-in-background recovery.
181 assert_eq!(
182 contract_bash_timeout_ms(true, None, false, false),
183 Some(CONTRACT_BASH_FOREGROUND_DEFAULT_TIMEOUT_MS)
184 );
185 const { assert!(CONTRACT_BASH_FOREGROUND_DEFAULT_TIMEOUT_MS < BASH_MAX_TIMEOUT_MS) };
186
187 // An explicit request still wins, including one far above the default:
188 // long foreground work stays possible when the model asks for it.
189 assert_eq!(
190 contract_bash_timeout_ms(true, Some(1_800_000), false, false),
191 Some(1_800_000)
192 );
193 assert_eq!(
194 contract_bash_timeout_ms(true, Some(5), false, false),
195 Some(5)
196 );
197
198 // Background and interactive runs are meant to outlive the call, so they
199 // keep "no timeout" and are never bounded by the foreground default.
200 assert_eq!(contract_bash_timeout_ms(true, None, true, false), None);
201 assert_eq!(contract_bash_timeout_ms(true, None, false, true), None);
202
203 // The standalone Bash tools already resolve their own default upstream;
204 // this helper must not second-guess the value they pass in.
205 assert_eq!(
206 contract_bash_timeout_ms(false, Some(120_000), false, false),
207 Some(120_000)
208 );
209 }
210
211 #[cfg(all(unix, not(target_env = "ohos")))]
212 #[test]
213 fn inherited_interactive_terminal_fails_closed_before_spawn() {
214 let workspace = tempdir().expect("workspace");
215 let mut manager = ShellManager::new(workspace.path().to_path_buf());
216 let err = manager
217 .execute_interactive_with_policy_env("codew", None, 10_000, None, HashMap::new())
218 .expect_err("Unix inherited-terminal takeover must not spawn");
219 let message = err.to_string();
220 assert!(message.contains("foreground TTY ownership"), "{message}");
221 assert!(message.contains("background: true, tty: true"), "{message}");
222 assert!(message.contains("action: \"interact\""), "{message}");
223 assert!(message.contains("task_id"), "{message}");
224 assert!(message.contains("terminal/run"), "{message}");
225 assert!(message.contains("terminal/send"), "{message}");
226 }
227
228 #[cfg(all(unix, target_env = "ohos"))]
229 #[test]
230 fn inherited_interactive_terminal_offers_only_ohos_recovery_paths() {
231 let workspace = tempdir().expect("workspace");
232 let mut manager = ShellManager::new(workspace.path().to_path_buf());
233 let err = manager
234 .execute_interactive_with_policy_env("codew", None, 10_000, None, HashMap::new())
235 .expect_err("OHOS inherited-terminal takeover must not spawn");
236 let message = err.to_string();
237 assert!(message.contains("foreground TTY ownership"), "{message}");
238 assert!(message.contains("new terminal"), "{message}");
239 assert!(message.contains("omit `interactive: true`"), "{message}");
240 assert!(!message.contains("background: true"), "{message}");
241 assert!(!message.contains("terminal/run"), "{message}");
242 }
243
244 #[test]
245 fn contract_bash_nonzero_is_an_error_with_status_after_output() {
246 let error = finish_contract_bash_result(
247 ShellResult {
248 task_id: None,
249 status: ShellStatus::Failed,
250 exit_code: Some(7),
251 stdout: "before".to_string(),
252 stderr: String::new(),
253 duration_ms: 1,
254 stdout_len: 6,
255 stderr_len: 0,
256 stdout_omitted: 0,
257 stderr_omitted: 0,
258 stdout_truncated: false,
259 stderr_truncated: false,
260 sandboxed: false,
261 sandbox_type: None,
262 sandbox_denied: false,
263 },
264 None,
265 &ToolContext::new("."),
266 None,
267 )
268 .expect_err("nonzero must be a failed tool call");
269 assert!(
270 error
271 .to_string()
272 .ends_with("before\n\nCommand exited with code 7")
273 );
274 // The error keeps the metadata a success carries, so hooks still see the
275 // exit code and status of a failing command.
276 let metadata = error.metadata().expect("failure metadata");
277 assert_eq!(metadata["exit_code"], 7);
278 assert_eq!(metadata["status"], "Failed");
279 }
280
281 #[cfg(unix)]
282 #[tokio::test]
283 async fn lowercase_bash_returns_one_ordered_stream() {
284 let workspace = tempdir().expect("workspace");
285 let context = ToolContext::new(workspace.path());
286 let result = LowercaseBashTool
287 .execute(
288 json!({"command": "printf out-1; printf err-2 >&2; printf out-3"}),
289 &context,
290 )
291 .await
292 .expect("bash");
293 assert_eq!(result.content, "out-1err-2out-3");
294 }
295
296 /// #6689: the receipt is exact-or-absent, fits the hook bound after JSON
297 /// escaping, and reports how the run ended without inventing an exit code.
298 #[test]
299 fn execution_receipt_is_bounded_and_reports_truthful_state() {
300 let tmp = tempdir().unwrap();
301 let identity = |command: &str, cwd: &Path| ShellExecutionIdentity {
302 command: command.to_string(),
303 cwd: cwd.to_path_buf(),
304 };
305 let cwd = tmp.path().canonicalize().unwrap();
306 let mut result = failed_network_shell_result(
307 &"\u{1f40b}\n\"\u{1}".repeat(12_000),
308 &"err\n".repeat(12_000),
309 );
310 result.sandboxed = false;
311 result.sandbox_type = None;
312 result.stdout_truncated = true;
313 result.exit_code = None;
314 result.status = ShellStatus::Killed;
315
316 let receipt = shell_execution_receipt(&identity("printf hi", &cwd), &result, "separate")
317 .expect("receipt");
318 let encoded = serde_json::to_string(&receipt).unwrap();
319 assert!(encoded.len() <= crate::hooks::HOOK_EXECUTION_RECEIPT_MAX_BYTES);
320 assert_eq!(serde_json::from_str::<Value>(&encoded).unwrap(), receipt);
321 assert_eq!(receipt["schema_version"], 1);
322 assert_eq!(receipt["command"], "printf hi");
323 assert_eq!(receipt["cwd"], cwd.to_str().unwrap());
324 assert_eq!(receipt["state"], "interrupted");
325 assert!(receipt["exit_code"].is_null());
326 assert_eq!(receipt["stdout_truncated"], true);
327 assert_eq!(receipt["stderr_truncated"], true);
328 assert!(
329 receipt["stderr"]
330 .as_str()
331 .unwrap()
332 .contains("[receipt preview truncated]")
333 );
334
335 // A nonzero exit is a completed run, and a 64-bit code survives.
336 result.status = ShellStatus::Failed;
337 result.exit_code = Some(3_221_225_477);
338 let receipt = shell_execution_receipt(&identity("x", &cwd), &result, "combined").unwrap();
339 assert_eq!(receipt["state"], "completed");
340 assert_eq!(receipt["exit_code"], 3_221_225_477_i64);
341 assert_eq!(receipt["output_kind"], "combined");
342
343 result.status = ShellStatus::TimedOut;
344 result.exit_code = None;
345 let receipt = shell_execution_receipt(&identity("x", &cwd), &result, "separate").unwrap();
346 assert_eq!(receipt["state"], "interrupted");
347
348 // Absent, never truncated or guessed.
349 result.status = ShellStatus::Running;
350 assert!(shell_execution_receipt(&identity("x", &cwd), &result, "separate").is_none());
351 result.status = ShellStatus::Completed;
352 for command in ["", "bad\0command"] {
353 assert!(shell_execution_receipt(&identity(command, &cwd), &result, "separate").is_none());
354 }
355 let long = "x".repeat(EXECUTION_RECEIPT_IDENTITY_MAX_BYTES + 1);
356 assert!(shell_execution_receipt(&identity(&long, &cwd), &result, "separate").is_none());
357 let long_cwd = PathBuf::from(format!("/{long}"));
358 assert!(shell_execution_receipt(&identity("x", &long_cwd), &result, "separate").is_none());
359 assert!(
360 shell_execution_receipt(&identity("x", Path::new("relative")), &result, "separate")
361 .is_none()
362 );
363 result.sandboxed = true;
364 assert!(shell_execution_receipt(&identity("x", &cwd), &result, "separate").is_none());
365 }
366
367 /// #6689: a settled foreground run records the command and directory the
368 /// process manager spawned, on success and on failure, and background runs
369 /// carry no receipt. The workspace is opened through a symlink so the default
370 /// directory and an explicit `cwd` would otherwise be spelled differently.
371 #[cfg(unix)]
372 #[tokio::test]
373 async fn foreground_shell_results_carry_the_spawned_execution_receipt() {
374 let tmp = tempdir().unwrap();
375 let real = tmp.path().join("real");
376 std::fs::create_dir_all(real.join("child")).unwrap();
377 let workspace = tmp.path().join("link");
378 std::os::unix::fs::symlink(&real, &workspace).unwrap();
379 let mut context = ToolContext::new(workspace.clone())
380 .with_elevated_sandbox_policy(ExecutionSandboxPolicy::DangerFullAccess);
381 context.auto_approve = true;
382 let tool = BashTool::new("Bash");
383
384 // No completion hook registered: nothing reads a receipt, so none rides
385 // along in the metadata the Runtime API persists.
386 let unobserved = tool
387 .execute(json!({"command": "pwd"}), &context)
388 .await
389 .unwrap();
390 assert!(
391 unobserved
392 .metadata
393 .as_ref()
394 .unwrap()
395 .get("execution_receipt")
396 .is_none()
397 );
398
399 let hooks = crate::hooks::HookExecutor::new(
400 crate::hooks::HooksConfig {
401 enabled: true,
402 hooks: vec![crate::hooks::Hook::new(
403 crate::hooks::HookEvent::ToolCallAfter,
404 "true",
405 )],
406 ..crate::hooks::HooksConfig::default()
407 },
408 workspace.clone(),
409 );
410 context.runtime.hook_executor = Some(std::sync::Arc::new(hooks));
411 // Exact strings, not re-canonicalized: both spellings must already agree.
412 let real = real.canonicalize().unwrap();
413 let real_str = real.to_str().unwrap();
414 let child_str = real.join("child");
415 let child_str = child_str.to_str().unwrap();
416
417 let default_dir = tool
418 .execute(json!({"command": "pwd"}), &context)
419 .await
420 .unwrap();
421 let explicit_dir = tool
422 .execute(json!({"command": "pwd", "cwd": "."}), &context)
423 .await
424 .unwrap();
425 for result in [&default_dir, &explicit_dir] {
426 assert_eq!(
427 result.metadata.as_ref().unwrap()["execution_receipt"]["cwd"],
428 real_str
429 );
430 }
431
432 let command = "printf effective; printf diagnostic >&2; exit 7";
433 let result = tool
434 .execute(json!({"command": command, "cwd": "child"}), &context)
435 .await
436 .unwrap();
437 let receipt = &result.metadata.as_ref().unwrap()["execution_receipt"];
438 assert_eq!(receipt["command"], command);
439 assert_eq!(receipt["cwd"], child_str);
440 assert_eq!(receipt["stdout"], "effective");
441 assert_eq!(receipt["stderr"], "diagnostic");
442 assert_eq!(receipt["exit_code"], 7);
443 assert_eq!(receipt["state"], "completed");
444 assert_eq!(receipt["output_kind"], "separate");
445
446 let interrupted = tool
447 .execute(json!({"command": "kill -TERM $$"}), &context)
448 .await
449 .unwrap();
450 let receipt = &interrupted.metadata.as_ref().unwrap()["execution_receipt"];
451 assert_eq!(receipt["state"], "interrupted");
452 assert!(receipt["exit_code"].is_null());
453
454 let background = tool
455 .execute(
456 json!({"command": "printf background", "background": true}),
457 &context,
458 )
459 .await
460 .unwrap();
461 assert!(
462 background
463 .metadata
464 .as_ref()
465 .unwrap()
466 .get("execution_receipt")
467 .is_none()
468 );
469
470 // Lowercase `bash` shares one pipe: the preview is combined output, and a
471 // failing command's error still carries the receipt for hooks.
472 let result = LowercaseBashTool
473 .execute(
474 json!({"command": "printf merged; printf diagnostic >&2"}),
475 &context,
476 )
477 .await
478 .unwrap();
479 let receipt = &result.metadata.as_ref().unwrap()["execution_receipt"];
480 assert_eq!(receipt["output_kind"], "combined");
481 assert_eq!(receipt["stdout"], "mergeddiagnostic");
482 assert_eq!(receipt["stderr"], "");
483 assert_eq!(receipt["state"], "completed");
484 let error = LowercaseBashTool
485 .execute(json!({"command": "printf partial; exit 3"}), &context)
486 .await
487 .expect_err("nonzero exit is a failed bash call");
488 let receipt = &error.metadata().expect("failure metadata")["execution_receipt"];
489 assert_eq!(receipt["command"], "printf partial; exit 3");
490 assert_eq!(receipt["exit_code"], 3);
491 assert_eq!(receipt["state"], "completed");
492 assert_eq!(receipt["cwd"], real_str);
493 }
494
495 /// A post-run lookup of a retargeted workspace symlink would falsely name
496 /// the replacement directory. The receipt must keep the actual spawn path.
497 #[cfg(unix)]
498 #[tokio::test]
499 async fn execution_receipt_keeps_spawn_cwd_when_the_command_retargets_the_workspace() {
500 let tmp = tempdir().unwrap();
501 let real = tmp.path().join("real");
502 let other = tmp.path().join("other");
503 std::fs::create_dir(&real).unwrap();
504 std::fs::create_dir(&other).unwrap();
505 let workspace = tmp.path().join("link");
506 std::os::unix::fs::symlink(&real, &workspace).unwrap();
507 let mut context = ToolContext::new(workspace.clone())
508 .with_elevated_sandbox_policy(ExecutionSandboxPolicy::DangerFullAccess);
509 context.auto_approve = true;
510 context.runtime.hook_executor = Some(std::sync::Arc::new(crate::hooks::HookExecutor::new(
511 crate::hooks::HooksConfig {
512 enabled: true,
513 hooks: vec![crate::hooks::Hook::new(
514 crate::hooks::HookEvent::ToolCallAfter,
515 "true",
516 )],
517 ..crate::hooks::HooksConfig::default()
518 },
519 workspace.clone(),
520 )));
521 let command = "pwd -P; rm ../link; ln -s other ../link; pwd -P";
522 let result = BashTool::new("Bash")
523 .execute(json!({"command": command}), &context)
524 .await
525 .unwrap();
526 assert!(result.success);
527 let receipt = &result.metadata.as_ref().unwrap()["execution_receipt"];
528 let actual = real.canonicalize().unwrap();
529 let actual = actual.to_str().unwrap();
530 assert_eq!(receipt["command"], command);
531 assert_eq!(receipt["cwd"], actual);
532 assert_eq!(receipt["stdout"], format!("{actual}\n{actual}\n"));
533 assert_eq!(receipt["exit_code"], 0);
534 assert_eq!(
535 workspace.canonicalize().unwrap(),
536 other.canonicalize().unwrap()
537 );
538 }
539
540 #[cfg(unix)]
541 #[tokio::test]
542 async fn lowercase_bash_keeps_raw_command_under_readonly_policy() {
543 let workspace = tempdir().expect("workspace");
544 let context = ToolContext::new(workspace.path())
545 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
546 let result = LowercaseBashTool
547 .execute(json!({"command": "pwd"}), &context)
548 .await
549 .expect("read-only bash");
550 assert_eq!(
551 result.content.trim(),
552 workspace
553 .path()
554 .canonicalize()
555 .expect("canonical workspace")
556 .display()
557 .to_string()
558 );
559 }
560
561 #[tokio::test]
562 async fn lowercase_bash_readonly_refusal_names_work_mode() {
563 let workspace = tempdir().expect("workspace");
564 let context = ToolContext::new(workspace.path())
565 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
566 let error = LowercaseBashTool
567 .execute(json!({"command": "touch blocked-by-plan"}), &context)
568 .await
569 .expect_err("policy refusal is a typed denial");
570
571 assert!(
572 matches!(error, ToolError::PermissionDenied { .. }),
573 "{error}"
574 );
575 let message = error.to_string();
576 assert!(message.contains("Work mode (`/mode work`)"), "{message}");
577 assert!(!message.contains("Act mode"));
578 assert!(!workspace.path().join("blocked-by-plan").exists());
579 }
580
581 /// Regression for the wedge that took out the owner's own session under swap
582 /// exhaustion: the lowercase `bash` spill file could not be created (full temp
583 /// volume), every call — including `echo ok` — failed with the harness-internal
584 /// "Failed to create streaming shell output", and nothing recovered. Spill
585 /// failure must be soft: the command still runs, the tail is still returned,
586 /// the next call still works, and no job state leaks.
587 #[cfg(unix)]
588 #[tokio::test]
589 async fn lowercase_bash_survives_spill_file_failure_and_stays_usable() {
590 let workspace = tempdir().expect("workspace");
591 let context = ToolContext::new(workspace.path());
592 let missing_spill_dir = workspace.path().join("no-such-temp-volume");
593 assert!(!missing_spill_dir.exists());
594 context
595 .shell_manager
596 .lock()
597 .expect("shell manager")
598 .set_output_spill_dir_for_test(Some(missing_spill_dir.clone()));
599
600 // (ii) the command runs and (i) no harness-internal error leaks.
601 let first = LowercaseBashTool
602 .execute(json!({"command": echo_command("ok")}), &context)
603 .await
604 .expect("bash runs even when the spill file cannot be created");
605 assert!(first.success, "{}", first.content);
606 assert_eq!(first.content.trim(), "ok");
607 assert!(
608 !first
609 .content
610 .contains("Failed to create streaming shell output")
611 );
612
613 // The next call must work too — the whole point of the fix.
614 let second = LowercaseBashTool
615 .execute(json!({"command": echo_command("still-ok")}), &context)
616 .await
617 .expect("second bash call after a spill failure");
618 assert!(second.success, "{}", second.content);
619 assert_eq!(second.content.trim(), "still-ok");
620
621 // Output past the bound is still delivered, and the notice explains why the
622 // full-output path is missing instead of pointing at a file that was never
623 // written.
624 let long = LowercaseBashTool
625 .execute(
626 json!({"command": "i=0; while [ $i -lt 2100 ]; do echo line-$i; i=$((i+1)); done"}),
627 &context,
628 )
629 .await
630 .expect("long bash output without a spill file");
631 assert!(long.success, "{}", long.content);
632 assert!(long.content.contains("line-2099"));
633 assert!(
634 long.content.contains("Full output was not persisted:"),
635 "{}",
636 long.content
637 );
638 assert!(!long.content.contains("Full output: "));
639
640 // (iii) no leaked session state: nothing is still running or unowned-pending.
641 let mut manager = context.shell_manager.lock().expect("shell manager");
642 let running = manager
643 .list_jobs()
644 .into_iter()
645 .filter(|job| job.status == ShellStatus::Running)
646 .count();
647 assert_eq!(running, 0, "no shell job may be left running");
648 assert!(
649 !missing_spill_dir.exists(),
650 "fail-soft must not create the dir"
651 );
652 }
653
654 /// A spawn/stream failure caused by host exhaustion must reach the model as
655 /// an actionable message (cause chain + likely reason + retry), never as a
656 /// bare harness-internal context string.
657 #[test]
658 fn shell_execution_failure_names_resource_exhaustion_and_says_retry() {
659 let error = anyhow::Error::from(std::io::Error::from(std::io::ErrorKind::StorageFull))
660 .context("Failed to open PTY");
661 let message = shell_execution_failed_message(&error);
662 assert!(
663 message.starts_with("Shell execution failed: Failed to open PTY"),
664 "{message}"
665 );
666 assert!(
667 message.contains("Likely host resource exhaustion"),
668 "{message}"
669 );
670 assert!(message.contains("disk"), "{message}");
671 assert!(message.contains("retry"), "{message}");
672 assert!(message.contains("still usable"), "{message}");
673
674 #[cfg(unix)]
675 {
676 let error = anyhow::Error::from(std::io::Error::from_raw_os_error(libc::EMFILE))
677 .context("Failed to spawn PTY command: echo ok");
678 let message = shell_execution_failed_message(&error);
679 assert!(message.contains("file descriptors"), "{message}");
680 assert!(message.contains("echo ok"), "{message}");
681 }
682
683 let plain = anyhow::anyhow!("working directory does not exist");
684 let message = shell_execution_failed_message(&plain);
685 assert_eq!(
686 message,
687 "Shell execution failed: working directory does not exist"
688 );
689 }
690
691 #[tokio::test]
692 async fn lowercase_bash_timeout_uses_seconds_and_fails() {
693 let workspace = tempdir().expect("workspace");
694 let context = ToolContext::new(workspace.path());
695 let error = LowercaseBashTool
696 .execute(
697 json!({"command": sleep_command(2), "timeout": 0.01}),
698 &context,
699 )
700 .await
701 .expect_err("timeout must fail");
702 assert!(
703 error
704 .to_string()
705 .contains("Command timed out after 0.01 seconds"),
706 "{error}"
707 );
708 let metadata = error.metadata().expect("timeout metadata");
709 assert_eq!(metadata["status"], "TimedOut");
710 }
711
712 fn execute_shell(
713 manager: &mut ShellManager,
714 command: &str,
715 working_dir: Option<&str>,
716 timeout_ms: u64,
717 background: bool,
718 ) -> Result<ShellResult> {
719 manager.execute_with_options_env_for_session(
720 command,
721 working_dir,
722 timeout_ms,
723 background,
724 None,
725 false,
726 None,
727 HashMap::new(),
728 "workspace",
729 )
730 }
731
732 #[test]
733 fn deleted_saved_workspace_reports_path_and_recovery_before_spawn() {
734 let workspace = tempdir().expect("workspace");
735 let stale = workspace.path().join("deleted-session-workspace");
736 let mut manager = ShellManager::new(stale.clone());
737
738 let error = execute_shell(&mut manager, "echo should-not-run", None, 1_000, false)
739 .expect_err("missing saved workspace must fail before shell spawn");
740 let message = error.to_string();
741 assert!(message.contains("saved session workspace is unavailable"));
742 assert!(message.contains(&stale.display().to_string()));
743 assert!(message.contains("working_dir") || message.contains("cwd"));
744 assert!(message.contains("resume/fork"));
745 }
746
747 #[test]
748 fn explicit_missing_working_dir_is_not_misreported_as_session_corruption() {
749 let workspace = tempdir().expect("workspace");
750 let missing = workspace.path().join("explicit-missing");
751 let mut manager = ShellManager::new(workspace.path().to_path_buf());
752
753 let error = execute_shell(
754 &mut manager,
755 "echo should-not-run",
756 missing.to_str(),
757 1_000,
758 false,
759 )
760 .expect_err("missing explicit cwd must fail before shell spawn");
761 let message = error.to_string();
762 assert!(message.contains("requested working directory is unavailable"));
763 assert!(message.contains(&missing.display().to_string()));
764 assert!(!message.contains("saved session workspace"));
765 }
766
767 #[cfg(not(target_env = "ohos"))]
768 #[test]
769 fn pty_exit_status_preserves_high_windows_code_losslessly() {
770 let raw = 0xC000_0005;
771 let status = ShellExitStatus::from_pty(portable_pty::ExitStatus::with_exit_code(raw));
772
773 assert!(!status.success);
774 assert_eq!(status.code, Some(i64::from(raw)));
775 assert_eq!(
776 exit_code_label(status.code),
777 "exit code 3221225477 (0xC0000005)"
778 );
779 assert_eq!(exit_code_hex(status.code).as_deref(), Some("0xC0000005"));
780 }
781
782 #[cfg(not(target_env = "ohos"))]
783 #[test]
784 fn ordinary_pty_exit_status_keeps_concise_label() {
785 let status = ShellExitStatus::from_pty(portable_pty::ExitStatus::with_exit_code(127));
786
787 assert_eq!(status.code, Some(127));
788 assert_eq!(exit_code_label(status.code), "exit code 127");
789 assert_eq!(exit_code_hex(status.code), None);
790 }
791
792 #[cfg(windows)]
793 #[test]
794 fn std_windows_exit_status_reinterprets_signed_dword() {
795 assert_eq!(std_exit_code_i64(0xC000_0005_u32 as i32), 0xC000_0005);
796 }
797
798 #[cfg(windows)]
799 const JOB_OBJECT_QUERY_ACCESS: u32 = 0x0004;
800
801 #[cfg(windows)]
802 fn duplicate_job_without_terminate_access(job: WindowsJob) -> WindowsJob {
803 let process = unsafe { GetCurrentProcess() };
804 let mut limited_handle = HANDLE::default();
805
806 unsafe {
807 DuplicateHandle(
808 process,
809 job.handle,
810 process,
811 &mut limited_handle,
812 JOB_OBJECT_QUERY_ACCESS,
813 false,
814 DUPLICATE_HANDLE_OPTIONS(0),
815 )
816 .expect("duplicate job handle without terminate access");
817 }
818
819 drop(job);
820 WindowsJob {
821 handle: limited_handle,
822 }
823 }
824
825 fn echo_command(message: &str) -> String {
826 format!("echo {message}")
827 }
828
829 fn sleep_command(seconds: u64) -> String {
830 let dispatcher = crate::shell_dispatcher::global_dispatcher();
831 if dispatcher.kind().is_powershell() {
832 return format!("Start-Sleep -Seconds {seconds}");
833 }
834 #[cfg(windows)]
835 {
836 let ping_count = seconds.saturating_add(1);
837 format!("ping 127.0.0.1 -n {ping_count} > NUL")
838 }
839 #[cfg(not(windows))]
840 {
841 format!("sleep {seconds}")
842 }
843 }
844
845 fn sleep_then_echo_command(seconds: u64, message: &str) -> String {
846 let dispatcher = crate::shell_dispatcher::global_dispatcher();
847 if dispatcher.kind().is_powershell() {
848 return format!("Start-Sleep -Seconds {seconds}; echo {message}");
849 }
850 #[cfg(windows)]
851 {
852 let ping_count = seconds.saturating_add(1);
853 format!("ping 127.0.0.1 -n {ping_count} > NUL && echo {message}")
854 }
855 #[cfg(not(windows))]
856 {
857 format!("sleep {seconds} && echo {message}")
858 }
859 }
860
861 fn echo_stdin_command() -> String {
862 let dispatcher = crate::shell_dispatcher::global_dispatcher();
863 if dispatcher.kind().is_powershell() {
864 return "[Console]::In.ReadToEnd()".to_string();
865 }
866 #[cfg(windows)]
867 {
868 "more".to_string()
869 }
870 #[cfg(not(windows))]
871 {
872 "cat".to_string()
873 }
874 }
875
876 fn network_restricted_context(tmp: &std::path::Path) -> ToolContext {
877 ToolContext::new(tmp)
878 .with_elevated_sandbox_policy(ExecutionSandboxPolicy::WorkspaceWrite {
879 writable_roots: vec![tmp.to_path_buf()],
880 network_access: false,
881 exclude_tmpdir: false,
882 exclude_slash_tmp: false,
883 })
884 .with_shell_network_denied_hint(
885 "Shell command blocked: Plan mode runs shell commands in a network-restricted sandbox.",
886 )
887 }
888
889 fn failed_network_shell_result(stdout: &str, stderr: &str) -> ShellResult {
890 ShellResult {
891 task_id: None,
892 status: ShellStatus::Failed,
893 exit_code: Some(6),
894 stdout: stdout.to_string(),
895 stderr: stderr.to_string(),
896 duration_ms: 25,
897 stdout_len: stdout.len(),
898 stderr_len: stderr.len(),
899 stdout_omitted: 0,
900 stderr_omitted: 0,
901 stdout_truncated: false,
902 stderr_truncated: false,
903 sandboxed: true,
904 sandbox_type: Some("seatbelt".to_string()),
905 sandbox_denied: false,
906 }
907 }
908
909 #[cfg(unix)]
910 const SHELL_DESCENDANT_HELPER_ENV: &str = "CODEWHALE_SHELL_DESCENDANT_HELPER";
911 #[cfg(unix)]
912 const SHELL_DESCENDANT_PID_FILE_ENV: &str = "CODEWHALE_SHELL_DESCENDANT_PID_FILE";
913
914 #[cfg(unix)]
915 #[test]
916 fn shell_descendant_helper_process() {
917 if std::env::var(SHELL_DESCENDANT_HELPER_ENV).ok().as_deref() != Some("1") {
918 return;
919 }
920 let pid_file =
921 PathBuf::from(std::env::var(SHELL_DESCENDANT_PID_FILE_ENV).expect("descendant pid file"));
922 let mut child = Command::new("sleep")
923 .arg("30")
924 .spawn()
925 .expect("spawn cheap descendant");
926 std::fs::write(pid_file, child.id().to_string()).expect("write descendant pid");
927 std::thread::sleep(Duration::from_secs(30));
928 let _ = child.wait();
929 }
930
931 #[cfg(unix)]
932 fn wait_for_shell_pid_file(path: &Path) -> libc::pid_t {
933 let deadline = Instant::now() + Duration::from_secs(5);
934 loop {
935 if let Ok(raw) = std::fs::read_to_string(path)
936 && let Ok(pid) = raw.trim().parse()
937 {
938 return pid;
939 }
940 assert!(
941 Instant::now() < deadline,
942 "descendant pid file never appeared"
943 );
944 std::thread::sleep(Duration::from_millis(25));
945 }
946 }
947
948 #[cfg(unix)]
949 fn wait_for_shell_pid_exit(pid: libc::pid_t) -> bool {
950 let deadline = Instant::now() + Duration::from_secs(2);
951 loop {
952 if unsafe { libc::kill(pid, 0) } != 0
953 && std::io::Error::last_os_error().raw_os_error() == Some(libc::ESRCH)
954 {
955 return true;
956 }
957 if Instant::now() >= deadline {
958 return false;
959 }
960 std::thread::sleep(Duration::from_millis(25));
961 }
962 }
963
964 fn wait_for_completed_shell(manager: &mut ShellManager, task_id: &str) -> ShellResult {
965 let deadline = Instant::now() + Duration::from_millis(BACKGROUND_COMPLETION_WAIT_MS);
966
967 loop {
968 let result = manager
969 .get_output(task_id, true, 1_000)
970 .expect("get_output");
971 if result.status != ShellStatus::Running || Instant::now() >= deadline {
972 return result;
973 }
974 std::thread::sleep(Duration::from_millis(50));
975 }
976 }
977
978 #[test]
979 fn shell_owner_registers_before_spawn_and_silent_work_stays_live() {
980 let work = crate::work_graph::new_shared_work_runtime(
981 crate::tools::todo::new_shared_todo_list(),
982 crate::tools::plan::new_shared_plan_state(),
983 );
984 let lifecycle = ShellWorkLifecycle {
985 work: work.clone(),
986 session_id: "shell-session".to_string(),
987 };
988
989 {
990 let _guard = ShellSpawnIntentGuard::new(
991 Some(lifecycle.clone()),
992 "shell_spawn_failure",
993 "missing-program",
994 );
995 }
996 lifecycle
997 .register("shell_silent", "sleep 30")
998 .expect("register silent shell");
999 lifecycle
1000 .observe("shell_silent", &ShellStatus::Running, 1, 0)
1001 .expect("live owner observation");
1002 lifecycle
1003 .observe("shell_silent", &ShellStatus::Running, 2, 512)
1004 .expect("growing output observation");
1005
1006 let graph = work
1007 .capture(Some("shell-session"))
1008 .expect("capture")
1009 .expect("graph")
1010 .graph;
1011 let operation = |external: &str| {
1012 graph.nodes.iter().find(|node| {
1013 node.binding
1014 .as_ref()
1015 .is_some_and(|binding| binding.external == external)
1016 })
1017 };
1018 assert_eq!(
1019 operation("shell:shell_spawn_failure").map(|node| node.state),
1020 Some(crate::work_graph::NodeState::Failed),
1021 "dropping an armed spawn guard must terminalize pre-spawn failure"
1022 );
1023 let silent = operation("shell:shell_silent").expect("silent shell operation");
1024 assert_eq!(silent.state, crate::work_graph::NodeState::Active);
1025 let observation = silent
1026 .binding
1027 .as_ref()
1028 .and_then(|binding| binding.last_observation.as_ref())
1029 .expect("last shell observation");
1030 assert_eq!(observation.seq, 2);
1031 assert_eq!(
1032 observation
1033 .output
1034 .as_ref()
1035 .and_then(crate::work_graph::EvidenceRef::raw_bytes),
1036 Some(512)
1037 );
1038 }
1039
1040 #[test]
1041 fn exec_shell_parallel_flags_are_input_aware() {
1042 let tool = BashTool::new("Bash");
1043 let readonly = json!({"command": "git status -s"});
1044 assert!(tool.supports_parallel_for(&readonly));
1045 assert!(tool.is_read_only_for(&readonly));
1046 assert_eq!(
1047 tool.approval_requirement_for(&readonly),
1048 ApprovalRequirement::Auto
1049 );
1050
1051 for input in [
1052 json!({"command": "fd -e rs ."}),
1053 json!({"command": "fd -H --type f src"}),
1054 json!({"command": "git grep TODO crates/tui/src/tools"}),
1055 json!({"action": "run", "command": "gh issue list --limit 10"}),
1056 json!({"action": "run", "command": "gh issue view 5287"}),
1057 ] {
1058 assert!(tool.supports_parallel_for(&input), "{input:?}");
1059 assert!(tool.is_read_only_for(&input), "{input:?}");
1060 assert_eq!(
1061 tool.approval_requirement_for(&input),
1062 ApprovalRequirement::Auto,
1063 "{input:?}"
1064 );
1065 }
1066
1067 for input in [
1068 json!({"command": "git status -s", "background": true}),
1069 json!({"command": "git status -s", "background": "false"}),
1070 json!({"command": "git status -s", "stdin": ""}),
1071 json!({"action": "wait", "command": "pwd", "task_id": "shell_1"}),
1072 json!({"action": "interact", "command": "pwd", "task_id": "shell_1"}),
1073 json!({"action": "cancel", "command": "pwd", "task_id": "shell_1"}),
1074 json!({"action": 3, "command": "pwd"}),
1075 json!({"command": "pwd", "unexpected": true}),
1076 json!({"command": "cargo build"}),
1077 json!({"command": "bash -lc 'git status'"}),
1078 json!({"command": "sh -c 'rg TODO crates'"}),
1079 json!({"command": "PAGER=./pwn.sh git log"}),
1080 json!({"command": "GH_PAGER=./pwn.sh gh issue view 5287"}),
1081 json!({"command": "rg ${9:---pre=./repo-script} needle ."}),
1082 json!({"command": "rg ${9:---hostname-bin=./repo-script} needle ."}),
1083 json!({"command": "fd ${9:---exec} ./repo-script"}),
1084 json!({"command": "rg $PATTERN ."}),
1085 json!({"command": "rg *.rs ."}),
1086 json!({"command": "bash -lc 'rg TODO crates | head'"}),
1087 json!({"command": "fd -x ./pwn.sh"}),
1088 json!({"command": "fd --exec ./pwn.sh"}),
1089 json!({"command": "fd -uHtx ./pwn.sh"}),
1090 json!({"command": "rg --pre /tmp/evil.sh needle ."}),
1091 json!({"command": "rg --hostname-bin ./repo-script --hyperlink-format=file://{host}{path} needle ."}),
1092 json!({"command": "rg --search-zip needle ."}),
1093 json!({"command": "rg -z needle ."}),
1094 json!({"command": "git grep -O needle"}),
1095 json!({"command": "git grep -nO needle"}),
1096 json!({"command": "git grep --textconv needle"}),
1097 json!({"command": "git diff --ext-diff HEAD"}),
1098 json!({"command": "git diff --textconv HEAD"}),
1099 json!({"command": "git log --show-signature -1"}),
1100 json!({"command": "git show --format=%GS HEAD"}),
1101 json!({"command": "gh issue close 5287"}),
1102 json!({"command": "gh issue view 5287 > issue.txt"}),
1103 json!({"command": "gh pr checks 42 --watch"}),
1104 json!({"command": "gh issue view 5287 -R git.example.com/o/r"}),
1105 ] {
1106 assert!(!tool.supports_parallel_for(&input), "{input:?}");
1107 assert!(!tool.is_read_only_for(&input), "{input:?}");
1108 assert_eq!(
1109 tool.approval_requirement_for(&input),
1110 ApprovalRequirement::Required,
1111 "{input:?}"
1112 );
1113 }
1114
1115 assert!(tool.starts_detached_for(&json!({
1116 "command": "cargo check --workspace",
1117 "background": true
1118 })));
1119 assert!(tool.starts_detached_for(&json!({
1120 "command": "cargo test -p codewhale-tui --bins",
1121 "tty": true
1122 })));
1123 assert!(!tool.starts_detached_for(&json!({
1124 "command": "cargo check --workspace"
1125 })));
1126 assert!(!tool.starts_detached_for(&json!({
1127 "command": "cargo check --workspace",
1128 "background": true,
1129 "interactive": true
1130 })));
1131 }
1132
1133 #[tokio::test]
1134 async fn readonly_shell_refuses_raw_string_external_backend() {
1135 struct Backend(std::sync::atomic::AtomicBool);
1136 #[async_trait::async_trait]
1137 impl crate::sandbox::backend::SandboxBackend for Backend {
1138 fn kind(&self) -> crate::sandbox::backend::SandboxKind {
1139 crate::sandbox::backend::SandboxKind::OpenSandbox
1140 }
1141 async fn exec(
1142 &self,
1143 _cmd: &str,
1144 _env: &std::collections::HashMap<String, String>,
1145 ) -> anyhow::Result<crate::sandbox::backend::SandboxOutput> {
1146 self.0.store(true, std::sync::atomic::Ordering::SeqCst);
1147 Ok(crate::sandbox::backend::SandboxOutput {
1148 stdout: String::new(),
1149 stderr: String::new(),
1150 exit_code: 0,
1151 })
1152 }
1153 }
1154
1155 let tmp = tempdir().expect("tempdir");
1156 let backend = std::sync::Arc::new(Backend(std::sync::atomic::AtomicBool::new(false)));
1157 let mut context = ToolContext::new(tmp.path().to_path_buf())
1158 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1159 context.sandbox_backend = Some(backend.clone());
1160 let error = BashTool::read_only("Bash")
1161 .execute(json!({"action": "run", "command": "pwd"}), &context)
1162 .await
1163 .expect_err("raw-string backend must not receive a classifier-approved argv")
1164 .to_string();
1165 assert!(error.contains("raw command string"), "{error}");
1166 assert!(!backend.0.load(std::sync::atomic::Ordering::SeqCst));
1167 }
1168
1169 #[tokio::test]
1170 async fn lowercase_bash_refuses_non_streaming_external_backend() {
1171 struct Backend(std::sync::atomic::AtomicBool);
1172 #[async_trait::async_trait]
1173 impl crate::sandbox::backend::SandboxBackend for Backend {
1174 fn kind(&self) -> crate::sandbox::backend::SandboxKind {
1175 crate::sandbox::backend::SandboxKind::OpenSandbox
1176 }
1177 async fn exec(
1178 &self,
1179 _cmd: &str,
1180 _env: &std::collections::HashMap<String, String>,
1181 ) -> anyhow::Result<crate::sandbox::backend::SandboxOutput> {
1182 self.0.store(true, std::sync::atomic::Ordering::SeqCst);
1183 unreachable!("lowercase bash must fail before external dispatch")
1184 }
1185 }
1186
1187 let workspace = tempdir().expect("workspace");
1188 let backend = std::sync::Arc::new(Backend(std::sync::atomic::AtomicBool::new(false)));
1189 let mut context = ToolContext::new(workspace.path());
1190 context.sandbox_backend = Some(backend.clone());
1191 let error = LowercaseBashTool
1192 .execute(json!({"command": "pwd", "timeout": 1}), &context)
1193 .await
1194 .expect_err("non-streaming backend must be rejected");
1195 assert!(error.to_string().contains("combined streaming output"));
1196 assert!(!backend.0.load(std::sync::atomic::Ordering::SeqCst));
1197 }
1198
1199 #[test]
1200 fn readonly_argv_is_shell_free_and_disables_git_helpers() {
1201 let (program, args) = hardened_readonly_argv("git show HEAD").expect("argv");
1202 assert_eq!(program, "git");
1203 assert_eq!(
1204 &args[..6],
1205 [
1206 "show",
1207 "--no-ext-diff",
1208 "--no-textconv",
1209 "--submodule=short",
1210 "--ignore-submodules=dirty",
1211 "--no-show-signature"
1212 ]
1213 );
1214 assert_eq!(args.last().map(String::as_str), Some("HEAD"));
1215
1216 let (_, args) = hardened_readonly_argv("git -C sub blame f.txt").expect("blame argv");
1217 assert_eq!(args, ["-C", "sub", "blame", "--no-textconv", "f.txt"]);
1218 assert_eq!(
1219 readonly_git_dirs(
1220 "git -C sub diff | git --no-pager -C /abs status | cat",
1221 std::path::Path::new("/ws")
1222 ),
1223 [
1224 std::path::PathBuf::from("/ws/sub"),
1225 std::path::PathBuf::from("/abs")
1226 ]
1227 );
1228 // Chains are split like the classifier splits them, so a `git` read
1229 // after `&&`, `||` or `;` still gets filter overrides.
1230 assert_eq!(
1231 readonly_git_dirs(
1232 "pwd && git diff; ls || git -C sub status",
1233 std::path::Path::new("/ws")
1234 ),
1235 [
1236 std::path::PathBuf::from("/ws"),
1237 std::path::PathBuf::from("/ws/sub")
1238 ]
1239 );
1240
1241 let (program, args) = hardened_readonly_argv("rg $PATTERN .").expect("literal argv");
1242 assert_eq!(program, "rg");
1243 assert_eq!(args, ["$PATTERN", "."]);
1244 }
1245
1246 /// Classifier-approved working-tree reads must not run the repository's clean
1247 /// filter or textconv driver. The helpers print a sentinel instead of the
1248 /// content, so this holds whether or not a kernel sandbox blocks their writes.
1249 #[cfg(unix)]
1250 #[tokio::test]
1251 async fn readonly_git_reads_run_no_clean_filter_or_textconv() {
1252 use crate::dependencies::ExternalTool as _;
1253 use std::os::unix::fs::PermissionsExt as _;
1254 let workspace = tempdir().expect("workspace");
1255 let outside = tempdir().expect("outside");
1256 let helper = outside.path().join("helper.sh");
1257 std::fs::write(&helper, "#!/bin/sh\necho HELPER-RAN\n").expect("helper");
1258 std::fs::set_permissions(&helper, std::fs::Permissions::from_mode(0o755)).expect("chmod");
1259 let helper = helper.display().to_string();
1260 let repo = workspace.path();
1261 let git = |args: &[&str]| {
1262 let status = crate::dependencies::Git::status(args, repo).expect("git should spawn");
1263 assert!(status.success(), "git {args:?} failed");
1264 };
1265 git(&["init", "-q"]);
1266 git(&["config", "user.email", "t@example.com"]);
1267 git(&["config", "user.name", "Test"]);
1268 git(&["config", "commit.gpgsign", "false"]);
1269 std::fs::write(repo.join(".gitattributes"), "c.txt filter=x diff=conv\n").expect("attrs");
1270 std::fs::write(repo.join("c.txt"), "c1\n").expect("write");
1271 git(&["add", "."]);
1272 git(&["commit", "-q", "-m", "init"]);
1273 git(&["config", "filter.x.clean", &helper]);
1274 git(&["config", "diff.conv.textconv", &helper]);
1275 std::fs::write(repo.join("c.txt"), "c2\n").expect("modify");
1276
1277 let ctx =
1278 ToolContext::new(repo).with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1279 let tool = BashTool::new("Bash");
1280 for command in ["git diff", "git blame c.txt"] {
1281 let result = tool
1282 .execute(json!({"command": command}), &ctx)
1283 .await
1284 .expect(command);
1285 assert!(result.success, "{command}: {}", result.content);
1286 assert!(
1287 !result.content.contains("HELPER-RAN"),
1288 "{command} ran a repository-configured command: {}",
1289 result.content
1290 );
1291 assert!(
1292 result.content.contains("c2"),
1293 "{command}: {}",
1294 result.content
1295 );
1296 }
1297 }
1298
1299 #[cfg(any(unix, windows))]
1300 #[test]
1301 fn readonly_program_resolution_ignores_workspace_shadow_executables() {
1302 let workspace = tempdir().expect("workspace");
1303 let trusted = tempdir().expect("trusted bin");
1304 let path = std::env::join_paths([workspace.path(), trusted.path()]).expect("test PATH");
1305
1306 for program in ["git", "gh", "rg"] {
1307 let file = if cfg!(windows) {
1308 format!("{program}.exe")
1309 } else {
1310 program.to_string()
1311 };
1312 for directory in [workspace.path(), trusted.path()] {
1313 let executable = directory.join(&file);
1314 std::fs::write(&executable, b"fixture").expect("fixture executable");
1315 #[cfg(unix)]
1316 {
1317 use std::os::unix::fs::PermissionsExt as _;
1318 let mut permissions = executable.metadata().unwrap().permissions();
1319 permissions.set_mode(0o755);
1320 std::fs::set_permissions(&executable, permissions).unwrap();
1321 }
1322 }
1323 let resolved =
1324 resolve_readonly_program_from_path(program, workspace.path(), &path).expect("resolved");
1325 assert_eq!(resolved, trusted.path().join(file).canonicalize().unwrap());
1326 assert!(resolved.is_absolute() && !resolved.starts_with(workspace.path()));
1327 }
1328 }
1329
1330 #[test]
1331 fn readonly_child_env_removes_git_and_github_redirects() {
1332 let mut command = std::process::Command::new("unused");
1333 let redirects = [
1334 "GIT_DIR",
1335 "GIT_COMMON_DIR",
1336 "GIT_EXEC_PATH",
1337 "GIT_OBJECT_DIRECTORY",
1338 "GIT_SSH_COMMAND",
1339 "GH_CONFIG_DIR",
1340 "GH_OTHER_PATH",
1341 ];
1342 for key in redirects {
1343 command.env(key, "outside");
1344 }
1345 command.env(READONLY_ENV_MARKER, "1");
1346 let env = HashMap::from([(READONLY_ENV_MARKER.to_string(), "1".to_string())]);
1347 remove_readonly_redirect_env(&mut command, &env);
1348 for key in redirects {
1349 assert!(
1350 command
1351 .get_envs()
1352 .any(|(name, value)| name == std::ffi::OsStr::new(key) && value.is_none()),
1353 "{key} must be removed from the child environment"
1354 );
1355 }
1356 assert!(
1357 command
1358 .get_envs()
1359 .any(|(name, value)| name == READONLY_ENV_MARKER && value.is_none())
1360 );
1361 }
1362
1363 #[test]
1364 fn readonly_operands_are_workspace_bounded_and_symlink_aware() {
1365 let workspace = tempdir().expect("workspace");
1366 let outside = tempdir().expect("outside");
1367 std::fs::write(workspace.path().join("inside.txt"), "inside").expect("inside file");
1368 std::fs::write(outside.path().join("secret.txt"), "secret").expect("outside file");
1369
1370 enforce_readonly_workspace_operands("cat inside.txt", workspace.path(), workspace.path())
1371 .expect("in-workspace operand");
1372 let inside_absolute = workspace
1373 .path()
1374 .join("inside.txt")
1375 .canonicalize()
1376 .expect("canonical inside file");
1377 enforce_readonly_workspace_operands(
1378 &format!("cat {}", inside_absolute.display()),
1379 workspace.path(),
1380 workspace.path(),
1381 )
1382 .expect("absolute in-workspace operand");
1383
1384 let outside_absolute = outside
1385 .path()
1386 .join("secret.txt")
1387 .canonicalize()
1388 .expect("canonical outside file");
1389 let error = enforce_readonly_workspace_operands(
1390 &format!("cat {}", outside_absolute.display()),
1391 workspace.path(),
1392 workspace.path(),
1393 )
1394 .expect_err("absolute outside operand must fail")
1395 .to_string();
1396 assert!(error.contains("operand.outside_workspace"), "{error}");
1397
1398 for command in [
1399 "cat ../secret.txt",
1400 "cat ~/.ssh/id_rsa",
1401 "cat /rooted-current-drive.txt",
1402 "cat C:secret",
1403 r"cat C:\secret",
1404 r"cat \\server\share\secret",
1405 ] {
1406 let error =
1407 enforce_readonly_workspace_operands(command, workspace.path(), workspace.path())
1408 .expect_err("out-of-workspace operand must fail")
1409 .to_string();
1410 assert!(error.contains("inside the workspace"), "{command}: {error}");
1411 }
1412
1413 #[cfg(unix)]
1414 {
1415 std::os::unix::fs::symlink(
1416 outside.path().join("secret.txt"),
1417 workspace.path().join("secret-link"),
1418 )
1419 .expect("outside symlink");
1420 let error = enforce_readonly_workspace_operands(
1421 "cat secret-link",
1422 workspace.path(),
1423 workspace.path(),
1424 )
1425 .expect_err("symlink escape must fail")
1426 .to_string();
1427 assert!(error.contains("resolves outside"), "{error}");
1428
1429 let subdir = workspace.path().join("subdir");
1430 std::fs::create_dir(&subdir).expect("subdir");
1431 std::os::unix::fs::symlink(
1432 outside.path().join("secret.txt"),
1433 subdir.join("secret-link"),
1434 )
1435 .expect("cwd-relative outside symlink");
1436 enforce_readonly_workspace_operands("cat secret-link", workspace.path(), &subdir)
1437 .expect_err("operands must resolve relative to the effective cwd");
1438 }
1439 }
1440
1441 #[test]
1442 fn windows_verbatim_and_drive_operands_survive_posix_split() {
1443 // `shell_words` splits with POSIX backslash-escaping, which silently eats
1444 // the separators of Windows absolute paths (`C:\Users\...` becomes
1445 // `C:Users...`) and mangles the `\\?\` verbatim prefix before operand
1446 // classification can see it. The protection doubles those backslashes so
1447 // the splitter round-trips the real path (the `\\?\` cases that the
1448 // verbatim strip alone could never reach).
1449 for (raw, expected) in [
1450 (r"\\?\C:\Users\foo\inside.txt", r"C:\Users\foo\inside.txt"),
1451 (r"C:\Users\foo\inside.txt", r"C:\Users\foo\inside.txt"),
1452 (r"\\server\share\secret", r"\\server\share\secret"),
1453 (r"\\.\device\path", r"\\.\device\path"),
1454 ] {
1455 let protected = normalize_windows_command_paths(&format!("cat {raw}"));
1456 let argv = shell_words::split(&protected).expect("split must succeed");
1457 assert_eq!(
1458 argv,
1459 vec!["cat".to_string(), expected.to_string()],
1460 "{raw} must survive the POSIX split"
1461 );
1462 }
1463 }
1464
1465 #[test]
1466 fn windows_path_protection_leaves_other_words_untouched() {
1467 // POSIX escapes, drive-relative spellings, and plain relative operands
1468 // are not Windows absolute paths and must round-trip unchanged.
1469 assert_eq!(
1470 normalize_windows_command_paths("echo a\\ b && cat inside.txt"),
1471 "echo a\\ b && cat inside.txt"
1472 );
1473 assert_eq!(
1474 normalize_windows_command_paths("cat C:secret"),
1475 "cat C:secret"
1476 );
1477 assert_eq!(
1478 normalize_windows_command_paths("cat inside.txt"),
1479 "cat inside.txt"
1480 );
1481 }
1482
1483 #[test]
1484 fn readonly_github_shell_calls_obey_the_host_network_policy_before_spawn() {
1485 let tmp = tempdir().expect("tempdir");
1486 let context = |default| {
1487 ToolContext::new(tmp.path()).with_network_policy(
1488 crate::network_policy::NetworkPolicyDecider::new(
1489 crate::network_policy::NetworkPolicy {
1490 default,
1491 ..crate::network_policy::NetworkPolicy::default()
1492 },
1493 None,
1494 ),
1495 )
1496 };
1497
1498 let allow = context(crate::network_policy::DecisionToml::Allow);
1499 enforce_readonly_network_reads("gh issue view 5287", &allow)
1500 .expect("allowed github.com policy");
1501
1502 let deny = context(crate::network_policy::DecisionToml::Deny);
1503 let denied = enforce_readonly_network_reads("gh issue list", &deny)
1504 .expect_err("deny must stop before spawning gh")
1505 .to_string();
1506 assert!(denied.contains("blocked by the active network policy"));
1507 enforce_readonly_network_reads("git status", &deny)
1508 .expect("local reads do not consult the network policy");
1509
1510 let prompt = context(crate::network_policy::DecisionToml::Prompt);
1511 let prompted = enforce_readonly_network_reads("gh issue view 5287", &prompt)
1512 .expect_err("headless Scout cannot prompt interactively")
1513 .to_string();
1514 assert!(prompted.contains("requires network approval"));
1515
1516 // Read-only agents: every admitted network segment is judged.
1517 let readonly_deny = context(crate::network_policy::DecisionToml::Deny)
1518 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1519 for command in ["gh pr view 1 | head", "ls && gh issue list"] {
1520 let denied = enforce_readonly_network_reads(command, &readonly_deny)
1521 .expect_err("a network read inside a composition is still judged")
1522 .to_string();
1523 assert!(denied.contains("blocked"), "{command}: {denied}");
1524 }
1525 // npm is refused by the command authority before a configured registry
1526 // could be mistaken for the fixed public-registry network label.
1527 let npm = json!({"command": "npm view x"});
1528 let refusal = exec_shell_input_agent_readonly_verdict(&npm).expect_err("npm needs approval");
1529 assert!(refusal.detail.contains("configuration"));
1530 // An ordinary full shell retains its existing policy/approval path.
1531 enforce_readonly_network_reads("npm view x", &deny)
1532 .expect("npm is not granted or refused by full-shell read-only detection");
1533 // A full shell keeps its historical scope: only a lone gh read.
1534 enforce_readonly_network_reads("gh pr view 1 | head", &deny)
1535 .expect("full shell pipelines are governed elsewhere");
1536 }
1537
1538 #[test]
1539 fn exec_shell_interact_requires_approval() {
1540 let tool = BashTool::alias("exec_shell_interact", "interact");
1541 assert_eq!(tool.approval_requirement(), ApprovalRequirement::Required);
1542 assert!(
1543 tool.capabilities()
1544 .contains(&ToolCapability::RequiresApproval)
1545 );
1546 }
1547
1548 #[tokio::test]
1549 async fn read_only_shell_policy_blocks_non_readonly_commands() {
1550 let tmp = tempdir().expect("tempdir");
1551 let ctx = ToolContext::new(tmp.path())
1552 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1553 let tool = BashTool::new("Bash");
1554
1555 for (input, rule) in [
1556 (json!({"command": "cargo build"}), "program:"),
1557 (
1558 json!({"command": "git status -s", "background": true}),
1559 "shape:",
1560 ),
1561 (
1562 json!({"command": "git --config-env=core.fsmonitor=SHELL status"}),
1563 "option:",
1564 ),
1565 (
1566 json!({"command": "git -cdiff.foo.textconv=./repo-script diff HEAD"}),
1567 "option:",
1568 ),
1569 (json!({"command": "rg -f/etc/passwd needle ."}), "option:"),
1570 (json!({"command": "touch x && ls"}), "program:"),
1571 ] {
1572 // A typed denial, so a Fleet worker's no-progress guard counts it.
1573 let error = tool
1574 .execute(input.clone(), &ctx)
1575 .await
1576 .expect_err("classifier refusal");
1577 assert!(
1578 matches!(error, ToolError::PermissionDenied { .. }),
1579 "{input}: {error}"
1580 );
1581 let message = error.to_string();
1582 assert!(message.contains("[shell.readonly.command]"), "{message}");
1583 assert!(message.contains(rule), "{input}: {message}");
1584 }
1585 assert!(!tmp.path().join("x").exists());
1586 }
1587
1588 #[tokio::test]
1589 async fn read_only_refusal_names_child_alternatives_instead_of_mode_switch() {
1590 // #6298: a child has no `/mode` to switch to — a refusal that tells it to
1591 // switch modes is a dead end beside an available absurd path. The child
1592 // branch must name the child's own alternatives and the escalation path,
1593 // and only tools a read-only child actually has (#6015).
1594 let tmp = tempdir().expect("tempdir");
1595 let child_ctx = ToolContext::new(tmp.path())
1596 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly)
1597 .with_owner_agent("agent_child", "child");
1598 let tool = BashTool::new("Bash");
1599 let message = tool
1600 .execute(json!({"command": "touch evil.txt"}), &child_ctx)
1601 .await
1602 .expect_err("refused")
1603 .to_string();
1604 let rejection = codewhale_execpolicy::command_safety::agent_readonly_verdict("touch evil.txt")
1605 .expect_err("touch is not a read");
1606 assert!(message.contains(&rejection.to_string()), "{message}");
1607 assert!(message.contains("program: `touch`"), "{message}");
1608 assert!(message.contains("File tool"), "{message}");
1609 assert!(
1610 message.contains("return your findings and the blocked probe to the parent"),
1611 "{message}"
1612 );
1613 for absent in ["/mode work", "Git", "Run tests", "merge_tree"] {
1614 assert!(!message.contains(absent), "{absent} in {message}");
1615 }
1616 assert!(!tmp.path().join("evil.txt").exists());
1617
1618 let parent_ctx = ToolContext::new(tmp.path())
1619 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1620 let message = tool
1621 .execute(json!({"command": "cargo build"}), &parent_ctx)
1622 .await
1623 .expect_err("refused")
1624 .to_string();
1625 assert!(message.contains("/mode work"), "{message}");
1626 }
1627
1628 #[test]
1629 fn leading_cd_moves_into_cwd_for_every_gate() {
1630 let rewritten = normalize_readonly_cd(&json!({"command": "cd sub && ls"}));
1631 assert_eq!(rewritten, json!({"command": "ls", "cwd": "sub"}));
1632 let rewritten =
1633 normalize_readonly_cd(&json!({"command": "cd inner && git diff", "cwd": "sub"}));
1634 assert_eq!(rewritten["command"], "git diff");
1635 assert_eq!(
1636 std::path::Path::new(rewritten["cwd"].as_str().unwrap()),
1637 std::path::Path::new("sub").join("inner")
1638 );
1639 for command in ["cd a; ls", "ls && cd b && ls", "cd && ls", "cd $X && ls"] {
1640 let input = json!({"command": command});
1641 assert_eq!(normalize_readonly_cd(&input), input, "{command}");
1642 assert!(!agent_readonly_bash_input(&input), "{command}");
1643 }
1644 // The enforced lane runs its command unchanged.
1645 let enforced = json!({"command": "cd sub && ls", "read_only": true});
1646 assert_eq!(normalize_readonly_cd(&enforced), enforced);
1647 assert!(agent_readonly_bash_input(
1648 &json!({"command": "cd sub && ls"})
1649 ));
1650 assert!(agent_readonly_bash_input(
1651 &json!({"command": "cd sub && git diff && echo '=== FILES ===' && ls -la"})
1652 ));
1653 }
1654
1655 #[cfg(unix)]
1656 #[tokio::test]
1657 async fn read_only_shell_runs_chains_and_leading_cd() {
1658 let workspace = tempdir().expect("workspace");
1659 std::fs::create_dir(workspace.path().join("sub")).expect("sub");
1660 std::fs::write(workspace.path().join("sub").join("a.txt"), "alpha\n").expect("a");
1661 std::fs::write(workspace.path().join("b.txt"), "beta\n").expect("b");
1662 let ctx = ToolContext::new(workspace.path())
1663 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1664 let tool = BashTool::new("Bash");
1665 let result = tool
1666 .execute(json!({"command": "cd sub && ls"}), &ctx)
1667 .await
1668 .expect("cd is moved into cwd");
1669 assert!(result.success, "{}", result.content);
1670 assert!(result.content.contains("a.txt"), "{}", result.content);
1671 assert!(!result.content.contains("b.txt"), "{}", result.content);
1672
1673 let result = tool
1674 .execute(
1675 json!({"command": "cat b.txt && echo --- && cat sub/a.txt 2>/dev/null"}),
1676 &ctx,
1677 )
1678 .await
1679 .expect("chain of reads");
1680 if !result.success && result.content.contains("require bash or zsh") {
1681 return;
1682 }
1683 assert!(result.success, "{}", result.content);
1684 assert!(result.content.contains("beta"), "{}", result.content);
1685 assert!(result.content.contains("---"), "{}", result.content);
1686 assert!(result.content.contains("alpha"), "{}", result.content);
1687
1688 let error = tool
1689 .execute(json!({"command": "cd .. && ls"}), &ctx)
1690 .await
1691 .expect_err("a cd outside the workspace is refused")
1692 .to_string();
1693 assert!(
1694 error.contains("escapes workspace") || error.contains("outside_workspace"),
1695 "{error}"
1696 );
1697 }
1698
1699 #[cfg(unix)]
1700 #[tokio::test]
1701 async fn read_only_operand_checks_apply_to_every_segment() {
1702 let workspace = tempdir().expect("workspace");
1703 let ctx = ToolContext::new(workspace.path())
1704 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1705 let error = BashTool::new("Bash")
1706 .execute(json!({"command": "gh pr view 1 && cat /etc/passwd"}), &ctx)
1707 .await
1708 .expect_err("the gh exemption covers only the gh segment");
1709 assert!(
1710 error.to_string().contains("shell.readonly.operand"),
1711 "{error}"
1712 );
1713 }
1714
1715 #[cfg(unix)]
1716 #[tokio::test]
1717 async fn read_only_shell_resolves_operands_from_the_effective_cwd() {
1718 let workspace = tempdir().expect("workspace");
1719 let outside = tempdir().expect("outside");
1720 let subdir = workspace.path().join("subdir");
1721 std::fs::create_dir(&subdir).expect("subdir");
1722 std::fs::write(outside.path().join("secret"), "secret").expect("outside secret");
1723 std::os::unix::fs::symlink(outside.path().join("secret"), subdir.join("secret-link"))
1724 .expect("symlink");
1725 let ctx = ToolContext::new(workspace.path())
1726 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1727 let error = BashTool::new("Bash")
1728 .execute(
1729 json!({"action": "run", "command": "cat secret-link", "cwd": "subdir"}),
1730 &ctx,
1731 )
1732 .await
1733 .expect_err("cwd-relative symlink escape must fail before spawn")
1734 .to_string();
1735 assert!(error.contains("resolves outside"), "{error}");
1736 }
1737
1738 #[cfg(unix)]
1739 #[tokio::test]
1740 async fn read_only_shell_skips_shell_env_hooks() {
1741 let tmp = tempdir().expect("tempdir");
1742 let marker = tmp.path().join("hook-ran");
1743 let hook = crate::hooks::Hook::new(
1744 crate::hooks::HookEvent::ShellEnv,
1745 &format!("printf hit > '{}'", marker.display()),
1746 );
1747 let executor = crate::hooks::HookExecutor::new(
1748 crate::hooks::HooksConfig {
1749 enabled: true,
1750 hooks: vec![hook],
1751 ..crate::hooks::HooksConfig::default()
1752 },
1753 tmp.path().to_path_buf(),
1754 );
1755 let mut context = ToolContext::new(tmp.path())
1756 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1757 context.runtime.hook_executor = Some(std::sync::Arc::new(executor));
1758
1759 let result = BashTool::read_only("Bash")
1760 .execute(json!({"command": "pwd"}), &context)
1761 .await
1762 .expect("read-only inspection");
1763 assert!(result.success, "{}", result.content);
1764 assert!(!marker.exists(), "shell_env hook must not run for ReadOnly");
1765 }
1766
1767 #[tokio::test]
1768 async fn read_only_shell_policy_allows_readonly_inspection() {
1769 let tmp = tempdir().expect("tempdir");
1770 let ctx = ToolContext::new(tmp.path())
1771 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
1772
1773 let result = BashTool::new("Bash")
1774 .execute(json!({"command": "pwd"}), &ctx)
1775 .await
1776 .expect("execute");
1777
1778 assert!(
1779 result.success,
1780 "unexpected shell failure: {}",
1781 result.content
1782 );
1783 assert_eq!(
1784 result
1785 .metadata
1786 .as_ref()
1787 .and_then(|metadata| metadata.get("status"))
1788 .and_then(Value::as_str),
1789 Some("Completed")
1790 );
1791 }
1792
1793 #[tokio::test]
1794 async fn exec_shell_multiline_block_explains_allow_shell_boundary() {
1795 let tmp = tempdir().expect("tempdir");
1796 let ctx = ToolContext::new(tmp.path());
1797
1798 let result = BashTool::new("Bash")
1799 .execute(
1800 json!({"command": "python3 -c \"print(1)\nprint(2)\""}),
1801 &ctx,
1802 )
1803 .await
1804 .expect("execute");
1805
1806 assert!(!result.success);
1807 assert!(result.content.contains("Command contains multiple lines"));
1808 assert!(
1809 result
1810 .content
1811 .contains("allow_shell=true exposes shell tools"),
1812 "{}",
1813 result.content
1814 );
1815 assert!(
1816 result
1817 .content
1818 .contains("Write multiline scripts to a file first"),
1819 "{}",
1820 result.content
1821 );
1822 assert!(
1823 result.content.contains("task_shell_start"),
1824 "{}",
1825 result.content
1826 );
1827 }
1828
1829 #[test]
1830 fn exec_shell_wait_schema_defaults_to_blocking() {
1831 let schema = BashTool::alias("exec_shell_wait", "wait").input_schema();
1832 assert!(
1833 schema["properties"]["wait"]["description"]
1834 .as_str()
1835 .is_some_and(|description| description.contains("default: true"))
1836 );
1837 assert!(
1838 BashTool::alias("exec_shell_wait", "wait")
1839 .description()
1840 .contains("wait")
1841 );
1842 }
1843
1844 #[tokio::test]
1845 async fn exec_shell_wait_without_wait_arg_blocks_until_completion() {
1846 let tmp = tempdir().expect("tempdir");
1847 let ctx = ToolContext::new(tmp.path());
1848 let start_result = BashTool::new("Bash")
1849 .execute(
1850 json!({"command": sleep_command(1), "background": true}),
1851 &ctx,
1852 )
1853 .await
1854 .expect("start background");
1855 let task_id = start_result
1856 .metadata
1857 .as_ref()
1858 .and_then(|metadata| metadata.get("task_id"))
1859 .and_then(Value::as_str)
1860 .expect("task id")
1861 .to_string();
1862
1863 let wait_result = BashTool::new("Bash")
1864 .execute(
1865 json!({"action": "wait", "task_id": task_id, "timeout_ms": 5_000}),
1866 &ctx,
1867 )
1868 .await
1869 .expect("wait for completion");
1870
1871 assert_eq!(
1872 wait_result
1873 .metadata
1874 .as_ref()
1875 .and_then(|metadata| metadata.get("status"))
1876 .and_then(Value::as_str),
1877 Some("Completed")
1878 );
1879 }
1880
1881 #[tokio::test]
1882 async fn exec_shell_wait_false_returns_nonblocking_snapshot() {
1883 let tmp = tempdir().expect("tempdir");
1884 let ctx = ToolContext::new(tmp.path());
1885 let start_result = BashTool::new("Bash")
1886 .execute(
1887 json!({"command": sleep_command(2), "background": true}),
1888 &ctx,
1889 )
1890 .await
1891 .expect("start background");
1892 let task_id = start_result
1893 .metadata
1894 .as_ref()
1895 .and_then(|metadata| metadata.get("task_id"))
1896 .and_then(Value::as_str)
1897 .expect("task id")
1898 .to_string();
1899
1900 let started = Instant::now();
1901 let wait_result = BashTool::new("Bash")
1902 .execute(
1903 json!({"action": "wait", "task_id": task_id, "timeout_ms": 5_000, "wait": false}),
1904 &ctx,
1905 )
1906 .await
1907 .expect("poll snapshot");
1908
1909 assert!(
1910 started.elapsed() < Duration::from_millis(1_000),
1911 "wait=false should return a snapshot without blocking"
1912 );
1913 assert_eq!(
1914 wait_result
1915 .metadata
1916 .as_ref()
1917 .and_then(|metadata| metadata.get("status"))
1918 .and_then(Value::as_str),
1919 Some("Running")
1920 );
1921 }
1922
1923 #[tokio::test]
1924 async fn exec_shell_wait_without_wait_arg_returns_running_at_timeout() {
1925 let tmp = tempdir().expect("tempdir");
1926 let ctx = ToolContext::new(tmp.path());
1927 let start_result = BashTool::new("Bash")
1928 .execute(
1929 json!({"command": sleep_command(5), "background": true}),
1930 &ctx,
1931 )
1932 .await
1933 .expect("start background");
1934 let task_id = start_result
1935 .metadata
1936 .as_ref()
1937 .and_then(|metadata| metadata.get("task_id"))
1938 .and_then(Value::as_str)
1939 .expect("task id")
1940 .to_string();
1941
1942 let started = Instant::now();
1943 let result = BashTool::new("Bash")
1944 .execute(
1945 json!({"action": "wait", "task_id": task_id, "timeout_ms": 1_000}),
1946 &ctx,
1947 )
1948 .await
1949 .expect("bounded wait");
1950 assert!(started.elapsed() >= Duration::from_millis(900));
1951 assert!(started.elapsed() < Duration::from_secs(3));
1952 assert_eq!(
1953 result
1954 .metadata
1955 .as_ref()
1956 .and_then(|metadata| metadata.get("status"))
1957 .and_then(Value::as_str),
1958 Some("Running")
1959 );
1960
1961 BashTool::new("Bash")
1962 .execute(json!({"action": "cancel", "task_id": task_id}), &ctx)
1963 .await
1964 .expect("cancel background");
1965 }
1966
1967 #[tokio::test]
1968 async fn exec_shell_wait_many_until_any_returns_when_the_first_task_settles() {
1969 let tmp = tempdir().expect("tempdir");
1970 let ctx = ToolContext::new(tmp.path());
1971 let short = BashTool::new("Bash")
1972 .execute(
1973 json!({"command": sleep_command(1), "background": true}),
1974 &ctx,
1975 )
1976 .await
1977 .expect("start short");
1978 let long = BashTool::new("Bash")
1979 .execute(
1980 json!({"command": sleep_command(30), "background": true}),
1981 &ctx,
1982 )
1983 .await
1984 .expect("start long");
1985 let short_id = short
1986 .metadata
1987 .as_ref()
1988 .and_then(|m| m.get("task_id"))
1989 .and_then(Value::as_str)
1990 .expect("short id")
1991 .to_string();
1992 let long_id = long
1993 .metadata
1994 .as_ref()
1995 .and_then(|m| m.get("task_id"))
1996 .and_then(Value::as_str)
1997 .expect("long id")
1998 .to_string();
1999
2000 let started = Instant::now();
2001 let result = BashTool::new("Bash")
2002 .execute(
2003 json!({
2004 "action": "wait",
2005 "task_ids": [short_id, long_id],
2006 "until": "any",
2007 "timeout_ms": 8_000
2008 }),
2009 &ctx,
2010 )
2011 .await
2012 .expect("multi wait");
2013
2014 assert!(
2015 started.elapsed() < Duration::from_secs(5),
2016 "until=any must not wait for the long task"
2017 );
2018 let statuses = result
2019 .metadata
2020 .as_ref()
2021 .expect("metadata")
2022 .get("statuses")
2023 .expect("statuses");
2024 assert_eq!(statuses[short_id.as_str()], "Completed");
2025 assert_eq!(statuses[long_id.as_str()], "Running");
2026 assert_eq!(
2027 result.metadata.as_ref().and_then(|m| m.get("until")),
2028 Some(&json!("any"))
2029 );
2030 assert_eq!(
2031 result.metadata.as_ref().and_then(|m| m.get("timed_out")),
2032 Some(&json!(false))
2033 );
2034 }
2035
2036 #[tokio::test]
2037 async fn exec_shell_wait_many_all_reports_timeout_and_still_running() {
2038 let tmp = tempdir().expect("tempdir");
2039 let ctx = ToolContext::new(tmp.path());
2040 let first = BashTool::new("Bash")
2041 .execute(
2042 json!({"command": sleep_command(30), "background": true}),
2043 &ctx,
2044 )
2045 .await
2046 .expect("start first");
2047 let second = BashTool::new("Bash")
2048 .execute(
2049 json!({"command": sleep_command(30), "background": true}),
2050 &ctx,
2051 )
2052 .await
2053 .expect("start second");
2054 let first_id = first
2055 .metadata
2056 .as_ref()
2057 .and_then(|m| m.get("task_id"))
2058 .and_then(Value::as_str)
2059 .expect("first id")
2060 .to_string();
2061 let second_id = second
2062 .metadata
2063 .as_ref()
2064 .and_then(|m| m.get("task_id"))
2065 .and_then(Value::as_str)
2066 .expect("second id")
2067 .to_string();
2068
2069 let result = BashTool::new("Bash")
2070 .execute(
2071 json!({ "action": "wait", "task_ids": [first_id, second_id], "timeout_ms": 1_500 }),
2072 &ctx,
2073 )
2074 .await
2075 .expect("multi wait times out");
2076 assert_eq!(
2077 result.metadata.as_ref().and_then(|m| m.get("timed_out")),
2078 Some(&json!(true))
2079 );
2080 assert!(result.content.contains("still running"));
2081 let statuses = result
2082 .metadata
2083 .as_ref()
2084 .expect("metadata")
2085 .get("statuses")
2086 .expect("statuses");
2087 assert_eq!(statuses[first_id.as_str()], "Running");
2088 assert_eq!(statuses[second_id.as_str()], "Running");
2089 }
2090
2091 #[tokio::test]
2092 async fn exec_shell_wait_many_rejects_an_unknown_task_id() {
2093 let tmp = tempdir().expect("tempdir");
2094 let ctx = ToolContext::new(tmp.path());
2095 let err = BashTool::new("Bash")
2096 .execute(
2097 json!({ "action": "wait", "task_ids": ["nope-1", "nope-2"], "wait": false }),
2098 &ctx,
2099 )
2100 .await
2101 .expect_err("unknown ids must fail the same way as the single-task path");
2102 assert!(err.to_string().contains("nope-1"), "{err}");
2103 }
2104
2105 #[tokio::test]
2106 async fn background_start_advertises_task_status_completion() {
2107 let tmp = tempdir().expect("tempdir");
2108 let ctx = ToolContext::new(tmp.path());
2109 let result = BashTool::new("Bash")
2110 .execute(
2111 json!({"command": sleep_command(1), "background": true}),
2112 &ctx,
2113 )
2114 .await
2115 .expect("start background");
2116 assert!(result.content.contains("completion is delivered"));
2117 assert!(result.content.contains("session exits") && result.content.contains("persist=true"));
2118 let metadata = result.metadata.as_ref().expect("metadata");
2119 assert_eq!(
2120 metadata
2121 .get("auto_resume_on_completion")
2122 .and_then(Value::as_bool),
2123 Some(true)
2124 );
2125 assert_eq!(
2126 metadata.get("completion_surface").and_then(Value::as_str),
2127 Some("runtime_event_and_task_status")
2128 );
2129 assert_eq!(
2130 metadata.get("background_policy").and_then(Value::as_str),
2131 Some("nonblocking")
2132 );
2133 }
2134
2135 #[tokio::test]
2136 async fn background_shell_job_preserves_origin_identity() {
2137 let tmp = tempdir().expect("tempdir");
2138 let ctx = ToolContext::new(tmp.path())
2139 .with_origin_turn_id("turn-origin")
2140 .with_origin_tool_call_id("tool-origin")
2141 .with_owner_agent("agent_owner", "verifier");
2142 let result = BashTool::new("Bash")
2143 .execute(
2144 json!({"command": sleep_command(2), "background": true}),
2145 &ctx,
2146 )
2147 .await
2148 .expect("start owned background shell");
2149
2150 let metadata = result.metadata.as_ref().expect("metadata");
2151 assert_eq!(
2152 metadata.get("owner_agent_id").and_then(Value::as_str),
2153 Some("agent_owner")
2154 );
2155 assert_eq!(
2156 metadata.get("owner_agent_name").and_then(Value::as_str),
2157 Some("verifier")
2158 );
2159 assert!(
2160 result
2161 .content
2162 .contains("not injected into the parent model"),
2163 "owned background work must describe its real completion route: {}",
2164 result.content
2165 );
2166 assert!(result.content.contains("Bash action=\"wait\""));
2167 assert_eq!(
2168 metadata
2169 .get("auto_resume_on_completion")
2170 .and_then(Value::as_bool),
2171 Some(false)
2172 );
2173 assert_eq!(
2174 metadata.get("completion_surface").and_then(Value::as_str),
2175 Some("task_status_and_explicit_wait")
2176 );
2177 let task_id = metadata
2178 .get("task_id")
2179 .and_then(Value::as_str)
2180 .expect("task id")
2181 .to_string();
2182
2183 {
2184 let mut manager = ctx.shell_manager.lock().expect("shell manager");
2185 let snapshot = manager
2186 .list_jobs()
2187 .into_iter()
2188 .find(|job| job.id == task_id)
2189 .expect("owned shell job snapshot");
2190 assert_eq!(snapshot.owner_agent_id.as_deref(), Some("agent_owner"));
2191 assert_eq!(snapshot.owner_agent_name.as_deref(), Some("verifier"));
2192 assert_eq!(snapshot.origin_tool_call_id.as_deref(), Some("tool-origin"));
2193 assert_eq!(snapshot.origin_turn_id.as_deref(), Some("turn-origin"));
2194 let mut legacy_json = serde_json::to_value(&snapshot).expect("serialize snapshot");
2195 let legacy_object = legacy_json.as_object_mut().expect("snapshot object");
2196 legacy_object.remove("origin_tool_call_id");
2197 legacy_object.remove("origin_turn_id");
2198 let legacy_snapshot: ShellJobSnapshot =
2199 serde_json::from_value(legacy_json).expect("deserialize legacy snapshot");
2200 assert_eq!(legacy_snapshot.origin_tool_call_id, None);
2201 assert_eq!(legacy_snapshot.origin_turn_id, None);
2202 let owners = manager.running_owner_agent_ids();
2203 assert_eq!(owners, vec!["agent_owner".to_string()]);
2204 }
2205
2206 BashTool::alias("exec_shell_cancel", "cancel")
2207 .execute(json!({"task_id": task_id}), &ctx)
2208 .await
2209 .expect("cancel owned background shell");
2210 }
2211
2212 #[tokio::test]
2213 async fn drain_finished_jobs_reports_once() {
2214 let tmp = tempdir().expect("tempdir");
2215 let ctx = ToolContext::new(tmp.path())
2216 .with_origin_turn_id("turn-origin")
2217 .with_origin_tool_call_id("tool-origin");
2218 let result = BashTool::new("Bash")
2219 .execute(
2220 json!({"command": echo_command("drain-finished-once"), "background": true}),
2221 &ctx,
2222 )
2223 .await
2224 .expect("start background");
2225 let task_id = result
2226 .metadata
2227 .as_ref()
2228 .and_then(|metadata| metadata.get("task_id"))
2229 .and_then(Value::as_str)
2230 .expect("task id")
2231 .to_string();
2232
2233 let mut manager = ctx.shell_manager.lock().expect("shell manager");
2234 assert!(manager.may_have_undelivered_completion());
2235 assert!(
2236 manager.may_have_undelivered_completion(),
2237 "read-only detection must not consume the pending completion"
2238 );
2239 let completed = wait_for_completed_shell(&mut manager, &task_id);
2240 assert_ne!(completed.status, ShellStatus::Running);
2241 assert!(manager.may_have_undelivered_completion());
2242
2243 let first = manager
2244 .drain_finished_jobs_with_evidence()
2245 .into_iter()
2246 .map(|completion| completion.event)
2247 .collect::<Vec<_>>();
2248 assert_eq!(first.len(), 1);
2249 assert_eq!(first[0].task_id, task_id);
2250 assert_eq!(first[0].status, ShellStatus::Completed);
2251 assert!(first[0].stdout_tail.contains("drain-finished-once"));
2252 assert_eq!(first[0].origin_tool_call_id.as_deref(), Some("tool-origin"));
2253 assert_eq!(first[0].origin_turn_id.as_deref(), Some("turn-origin"));
2254 let mut legacy_json = serde_json::to_value(&first[0]).expect("serialize completion");
2255 let legacy_object = legacy_json.as_object_mut().expect("completion object");
2256 legacy_object.remove("origin_tool_call_id");
2257 legacy_object.remove("origin_turn_id");
2258 let legacy_completion: ShellCompletionEvent =
2259 serde_json::from_value(legacy_json).expect("deserialize legacy completion");
2260 assert_eq!(legacy_completion.origin_tool_call_id, None);
2261 assert_eq!(legacy_completion.origin_turn_id, None);
2262
2263 let second = manager.drain_finished_jobs_with_evidence();
2264 assert!(second.is_empty(), "completion should be reported only once");
2265 assert!(!manager.may_have_undelivered_completion());
2266 }
2267
2268 #[tokio::test]
2269 async fn background_job_is_hidden_from_replacement_session_and_resumes_once_for_owner() {
2270 let tmp = tempdir().expect("tempdir");
2271 let ctx_a = ToolContext::new(tmp.path()).with_state_namespace("session-a");
2272 let ctx_b = ctx_a.clone().with_state_namespace("session-b");
2273 let result = BashTool::new("Bash")
2274 .execute(
2275 json!({"command": echo_command("owned-by-a"), "background": true}),
2276 &ctx_a,
2277 )
2278 .await
2279 .expect("start A background job");
2280 let task_id = result
2281 .metadata
2282 .as_ref()
2283 .and_then(|metadata| metadata.get("task_id"))
2284 .and_then(Value::as_str)
2285 .expect("task id")
2286 .to_string();
2287
2288 let mut manager = ctx_b.shell_manager.lock().expect("shell manager");
2289 assert!(manager.list_jobs_for_session("session-b").is_empty());
2290 assert!(
2291 manager
2292 .inspect_job_for_session("session-b", &task_id)
2293 .is_err()
2294 );
2295 assert!(
2296 manager
2297 .write_stdin_for_session("session-b", &task_id, "foreign", false)
2298 .is_err()
2299 );
2300 assert!(manager.kill_for_session("session-b", &task_id).is_err());
2301
2302 let completed = wait_for_completed_shell(&mut manager, &task_id);
2303 assert_ne!(completed.status, ShellStatus::Running);
2304 let owned = manager
2305 .list_jobs_for_session("session-a")
2306 .into_iter()
2307 .find(|job| job.id == task_id)
2308 .expect("A job remains visible to A");
2309 assert_eq!(owned.owner_session_id, "session-a");
2310 assert!(
2311 manager
2312 .drain_finished_jobs_with_evidence_for_session("session-b")
2313 .is_empty(),
2314 "B must not claim A's completion"
2315 );
2316 assert!(manager.has_finished_unreported_jobs_for_session("session-a"));
2317 let first = manager.drain_finished_jobs_with_evidence_for_session("session-a");
2318 assert_eq!(first.len(), 1);
2319 assert_eq!(first[0].event.owner_session_id, "session-a");
2320 assert!(first[0].event.stdout_tail.contains("owned-by-a"));
2321 assert!(
2322 manager
2323 .drain_finished_jobs_with_evidence_for_session("session-a")
2324 .is_empty(),
2325 "A completion is delivered exactly once"
2326 );
2327 }
2328
2329 #[test]
2330 fn completion_evidence_preserves_arbitrary_stream_bytes() {
2331 use base64::Engine as _;
2332
2333 let stdout = vec![b'o', 0, 0xff, b'k'];
2334 let stderr = vec![0xfe, b'e', b'r', b'r'];
2335 let evidence = ShellCompletionEvidence {
2336 event: ShellCompletionEvent {
2337 task_id: "shell_binary".to_string(),
2338 command: "binary-output".to_string(),
2339 status: ShellStatus::Completed,
2340 exit_code: Some(0),
2341 duration_ms: 17,
2342 stdout_tail: String::new(),
2343 stderr_tail: String::new(),
2344 stdout_len: stdout.len(),
2345 stderr_len: stderr.len(),
2346 evidence_ref: None,
2347 linked_task_id: None,
2348 owner_agent_id: None,
2349 owner_agent_name: None,
2350 origin_tool_call_id: Some("tool-origin".to_string()),
2351 origin_turn_id: Some("turn-origin".to_string()),
2352 owner_session_id: "session-test".to_string(),
2353 },
2354 stdout: stdout.clone(),
2355 stderr: stderr.clone(),
2356 stdout_omitted: 0,
2357 stderr_omitted: 0,
2358 };
2359
2360 let payload: serde_json::Value =
2361 serde_json::from_slice(&evidence.artifact_bytes()).expect("evidence JSON");
2362 assert_eq!(payload["stdout"]["encoding"], "base64");
2363 assert_eq!(payload["stderr"]["encoding"], "base64");
2364 assert_eq!(payload["origin_tool_call_id"], "tool-origin");
2365 assert_eq!(payload["origin_turn_id"], "turn-origin");
2366 let decoded_stdout = base64::engine::general_purpose::STANDARD
2367 .decode(payload["stdout"]["content"].as_str().expect("stdout data"))
2368 .expect("decode stdout");
2369 let decoded_stderr = base64::engine::general_purpose::STANDARD
2370 .decode(payload["stderr"]["content"].as_str().expect("stderr data"))
2371 .expect("decode stderr");
2372 assert_eq!(decoded_stdout, stdout);
2373 assert_eq!(decoded_stderr, stderr);
2374 }
2375
2376 #[test]
2377 #[cfg(unix)]
2378 fn shell_execution_scrubs_parent_env_and_keeps_explicit_env() {
2379 let _guard = env_lock().lock().expect("env lock");
2380 let previous = std::env::var_os("DEEPSEEK_CHILD_ENV_SHELL_SECRET");
2381 unsafe {
2382 std::env::set_var("DEEPSEEK_CHILD_ENV_SHELL_SECRET", "parent-secret");
2383 }
2384
2385 let tmp = tempdir().expect("tempdir");
2386 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2387 let mut extra = std::collections::HashMap::new();
2388 extra.insert(
2389 "DEEPSEEK_CHILD_ENV_EXPLICIT".to_string(),
2390 "explicit-value".to_string(),
2391 );
2392
2393 let result = manager
2394 .execute_with_options_env(
2395 "sh -c 'printf \"%s\\n%s\\n\" \"${DEEPSEEK_CHILD_ENV_SHELL_SECRET-unset}\" \"${DEEPSEEK_CHILD_ENV_EXPLICIT-unset}\"'",
2396 None,
2397 5000,
2398 false,
2399 None,
2400 false,
2401 None,
2402 extra,
2403 )
2404 .expect("execute");
2405
2406 match previous {
2407 Some(value) => unsafe {
2408 std::env::set_var("DEEPSEEK_CHILD_ENV_SHELL_SECRET", value);
2409 },
2410 None => unsafe {
2411 std::env::remove_var("DEEPSEEK_CHILD_ENV_SHELL_SECRET");
2412 },
2413 }
2414
2415 assert_eq!(result.status, ShellStatus::Completed);
2416 assert_eq!(result.stdout, "unset\nexplicit-value\n");
2417 }
2418
2419 #[test]
2420 #[cfg(windows)]
2421 fn shell_execution_preserves_custom_windows_sdk_root_env() {
2422 let _guard = env_lock().lock().expect("env lock");
2423 let previous_sdk = std::env::var_os("BIMRV_SDK_ROOT");
2424 let previous_secret = std::env::var_os("MY_SECRET_ROOT");
2425 unsafe {
2426 std::env::set_var("BIMRV_SDK_ROOT", r"F:\Lib\BimRv27.5");
2427 std::env::set_var("MY_SECRET_ROOT", r"F:\Secrets");
2428 }
2429
2430 let tmp = tempdir().expect("tempdir");
2431 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2432 let command = if crate::shell_dispatcher::global_dispatcher()
2433 .kind()
2434 .is_powershell()
2435 {
2436 r#"[Console]::WriteLine($env:BIMRV_SDK_ROOT); if ($null -eq $env:MY_SECRET_ROOT) { [Console]::WriteLine("secret-unset") } else { [Console]::WriteLine("secret-set") }"#
2437 .to_string()
2438 } else {
2439 r#"echo %BIMRV_SDK_ROOT% & if defined MY_SECRET_ROOT (echo secret-set) else (echo secret-unset)"#
2440 .to_string()
2441 };
2442
2443 let result = execute_shell(&mut manager, &command, None, 5000, false).expect("execute");
2444
2445 unsafe {
2446 match previous_sdk {
2447 Some(value) => std::env::set_var("BIMRV_SDK_ROOT", value),
2448 None => std::env::remove_var("BIMRV_SDK_ROOT"),
2449 }
2450 match previous_secret {
2451 Some(value) => std::env::set_var("MY_SECRET_ROOT", value),
2452 None => std::env::remove_var("MY_SECRET_ROOT"),
2453 }
2454 }
2455
2456 assert_eq!(result.status, ShellStatus::Completed);
2457 assert!(
2458 result.stdout.contains(r"F:\Lib\BimRv27.5"),
2459 "custom SDK root should reach exec_shell stdout: {:?}",
2460 result
2461 );
2462 assert!(
2463 result.stdout.contains("secret-unset"),
2464 "secret-like env should stay scrubbed: {:?}",
2465 result
2466 );
2467 }
2468
2469 #[test]
2470 fn test_sync_execution() {
2471 let tmp = tempdir().expect("tempdir");
2472 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2473
2474 let result =
2475 execute_shell(&mut manager, &echo_command("hello"), None, 5000, false).expect("execute");
2476
2477 assert_eq!(result.status, ShellStatus::Completed);
2478 assert!(result.stdout.contains("hello"));
2479 assert!(result.task_id.is_none());
2480 }
2481
2482 #[test]
2483 fn test_background_execution() {
2484 let tmp = tempdir().expect("tempdir");
2485 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2486
2487 let result = execute_shell(
2488 &mut manager,
2489 &sleep_then_echo_command(1, "done"),
2490 None,
2491 5000,
2492 true,
2493 )
2494 .expect("execute");
2495
2496 assert_eq!(result.status, ShellStatus::Running);
2497 assert!(result.task_id.is_some());
2498
2499 let task_id = result
2500 .task_id
2501 .expect("background execution should return task_id");
2502
2503 let final_result = wait_for_completed_shell(&mut manager, &task_id);
2504
2505 assert_eq!(final_result.status, ShellStatus::Completed);
2506 assert!(final_result.stdout.contains("done"));
2507 }
2508
2509 #[test]
2510 fn test_timeout() {
2511 let tmp = tempdir().expect("tempdir");
2512 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2513
2514 let result =
2515 execute_shell(&mut manager, &sleep_command(10), None, 1000, false).expect("execute");
2516
2517 assert_eq!(result.status, ShellStatus::TimedOut);
2518 }
2519
2520 #[test]
2521 fn test_kill() {
2522 let tmp = tempdir().expect("tempdir");
2523 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2524
2525 let result =
2526 execute_shell(&mut manager, &sleep_command(60), None, 5000, true).expect("execute");
2527
2528 let task_id = result
2529 .task_id
2530 .expect("background execution should return task_id");
2531
2532 // Kill it
2533 let killed = manager.kill(&task_id).expect("kill");
2534 assert_eq!(killed.status, ShellStatus::Killed);
2535 }
2536
2537 #[test]
2538 fn test_write_stdin_streams_output() {
2539 let tmp = tempdir().expect("tempdir");
2540 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2541
2542 let result = manager
2543 .execute_with_options_env(
2544 &echo_stdin_command(),
2545 None,
2546 5000,
2547 true,
2548 None,
2549 false,
2550 None,
2551 HashMap::new(),
2552 )
2553 .expect("execute");
2554
2555 let task_id = result
2556 .task_id
2557 .expect("background execution should return task_id");
2558
2559 manager
2560 .write_stdin(&task_id, "hello\n", true)
2561 .expect("write stdin");
2562
2563 let delta = manager
2564 .get_output_delta(&task_id, true, 5000)
2565 .expect("get_output_delta");
2566
2567 assert!(delta.result.stdout.contains("hello"));
2568
2569 let delta2 = manager
2570 .get_output_delta(&task_id, false, 0)
2571 .expect("get_output_delta");
2572 assert!(delta2.result.stdout.is_empty());
2573 }
2574
2575 #[test]
2576 #[cfg(all(unix, not(target_env = "ohos")))]
2577 fn background_tty_command_has_controlling_terminal() {
2578 let tmp = tempdir().expect("tempdir");
2579 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2580
2581 let result = manager
2582 .execute_with_options_env(
2583 "sh -c 'exec 3<>/dev/tty && printf tty-ok && exec 3>&-'",
2584 None,
2585 5000,
2586 true,
2587 None,
2588 true,
2589 Some(ExecutionSandboxPolicy::DangerFullAccess),
2590 HashMap::new(),
2591 )
2592 .expect("execute tty command");
2593
2594 let task_id = result
2595 .task_id
2596 .expect("background tty execution should return task_id");
2597
2598 let done = manager
2599 .get_output(&task_id, true, 10_000)
2600 .expect("get tty command output");
2601
2602 assert_eq!(done.status, ShellStatus::Completed);
2603 assert_eq!(done.exit_code, Some(0));
2604 assert!(
2605 done.stdout.contains("tty-ok"),
2606 "tty output should confirm /dev/tty opened; got {done:?}"
2607 );
2608 }
2609
2610 #[test]
2611 fn test_job_list_poll_cancel_and_stale_snapshot() {
2612 let tmp = tempdir().expect("tempdir");
2613 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2614
2615 let started = execute_shell(
2616 &mut manager,
2617 &sleep_then_echo_command(1, "done"),
2618 None,
2619 5000,
2620 true,
2621 )
2622 .expect("execute");
2623 let task_id = started.task_id.expect("task id");
2624 manager
2625 .tag_linked_task(&task_id, Some("task_123".to_string()))
2626 .expect("tag linked task");
2627
2628 let running = manager.list_jobs();
2629 let job = running
2630 .iter()
2631 .find(|job| job.id == task_id)
2632 .expect("running job");
2633 assert_eq!(job.status, ShellStatus::Running);
2634 assert_eq!(job.linked_task_id.as_deref(), Some("task_123"));
2635 assert!(job.command.contains("done"));
2636 assert_eq!(job.cwd, tmp.path());
2637
2638 let completed = manager
2639 .poll_delta(&task_id, true, 5000)
2640 .expect("poll delta");
2641 assert_eq!(completed.result.status, ShellStatus::Completed);
2642 assert!(completed.result.stdout.contains("done"));
2643
2644 let detail = manager.inspect_job(&task_id).expect("inspect");
2645 assert!(detail.stdout.contains("done"));
2646 assert_eq!(detail.snapshot.status, ShellStatus::Completed);
2647
2648 manager.remember_stale_job(
2649 "shell_stale",
2650 "cargo test",
2651 tmp.path().to_path_buf(),
2652 Some("task_old".to_string()),
2653 );
2654 let stale = manager
2655 .list_jobs()
2656 .into_iter()
2657 .find(|job| job.id == "shell_stale")
2658 .expect("stale job");
2659 assert!(stale.stale);
2660 assert_eq!(stale.linked_task_id.as_deref(), Some("task_old"));
2661 }
2662
2663 #[test]
2664 fn running_job_snapshot_marks_no_output_stale_after_threshold() {
2665 let tmp = tempdir().expect("tempdir");
2666 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2667
2668 let started =
2669 execute_shell(&mut manager, &sleep_command(5), None, 5000, true).expect("execute");
2670 let task_id = started.task_id.expect("task id");
2671
2672 {
2673 let shell = manager.processes.get_mut(&task_id).expect("live shell");
2674 shell.last_output_at = Instant::now() - STALE_NO_OUTPUT_AFTER - Duration::from_millis(1);
2675 }
2676
2677 let job = manager
2678 .list_jobs()
2679 .into_iter()
2680 .find(|job| job.id == task_id)
2681 .expect("running job");
2682
2683 assert_eq!(job.status, ShellStatus::Running);
2684 assert!(job.stale, "silent running job should be marked stale");
2685 assert!(
2686 job.elapsed_since_output_ms
2687 .is_some_and(|elapsed| elapsed >= STALE_NO_OUTPUT_AFTER.as_millis() as u64),
2688 "elapsed no-output time should be exposed: {job:?}"
2689 );
2690 }
2691
2692 #[test]
2693 fn running_job_snapshot_keeps_recent_no_output_fresh() {
2694 let tmp = tempdir().expect("tempdir");
2695 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2696
2697 let started =
2698 execute_shell(&mut manager, &sleep_command(5), None, 5000, true).expect("execute");
2699 let task_id = started.task_id.expect("task id");
2700
2701 let job = manager
2702 .list_jobs()
2703 .into_iter()
2704 .find(|job| job.id == task_id)
2705 .expect("running job");
2706
2707 assert_eq!(job.status, ShellStatus::Running);
2708 assert!(!job.stale, "fresh running job should not start stale");
2709 assert!(job.elapsed_since_output_ms.is_some());
2710 }
2711
2712 #[test]
2713 fn test_job_cancel_updates_completion_state() {
2714 let tmp = tempdir().expect("tempdir");
2715 let mut manager = ShellManager::new(tmp.path().to_path_buf());
2716
2717 let started =
2718 execute_shell(&mut manager, &sleep_command(60), None, 5000, true).expect("execute");
2719 let task_id = started.task_id.expect("task id");
2720
2721 let killed = manager.kill(&task_id).expect("kill");
2722 assert_eq!(killed.status, ShellStatus::Killed);
2723 let job = manager.inspect_job(&task_id).expect("inspect");
2724 assert_eq!(job.snapshot.status, ShellStatus::Killed);
2725 assert!(!job.snapshot.stdin_available);
2726 }
2727
2728 #[test]
2729 fn test_output_truncation() {
2730 let long_output = "x".repeat(50_000);
2731 let (truncated, _meta) = truncate_with_meta(&long_output);
2732
2733 assert!(truncated.len() < long_output.len());
2734 assert!(truncated.contains("truncated"));
2735 }
2736
2737 #[test]
2738 fn test_truncate_with_meta_reports_omission_counts() {
2739 let long_output = format!("line1\nline2\n{}", "x".repeat(60_000));
2740 let (truncated, meta) = truncate_with_meta(&long_output);
2741
2742 assert!(meta.truncated);
2743 assert!(meta.original_len >= long_output.len());
2744 assert!(meta.omitted > 0);
2745 assert!(truncated.contains("bytes omitted"));
2746 }
2747
2748 #[test]
2749 fn network_restricted_hint_detects_silent_curl_failure() {
2750 let tmp = tempdir().expect("tempdir");
2751 let ctx = network_restricted_context(tmp.path());
2752 let result = failed_network_shell_result("000", "");
2753
2754 let hint = shell_network_restricted_hint(
2755 &ctx,
2756 "curl -s -o /dev/null -w '%{http_code}' https://api.github.com",
2757 &result,
2758 )
2759 .expect("network-restricted hint");
2760
2761 assert!(hint.contains("Plan mode"));
2762 }
2763
2764 #[test]
2765 fn sandbox_denied_hint_names_the_effective_posture() {
2766 // DGF-02: an approved write blocked by a read-only sandbox must come
2767 // back naming the sandbox as the blocker, never as a bare failure.
2768 let tmp = tempdir().expect("tempdir");
2769 let ctx =
2770 ToolContext::new(tmp.path()).with_elevated_sandbox_policy(ExecutionSandboxPolicy::ReadOnly);
2771 let mut result =
2772 failed_network_shell_result("", "sh: cannot create out.txt: Operation not permitted");
2773 result.sandbox_denied = true;
2774
2775 let hint = shell_sandbox_denied_hint(&ctx, &result).expect("sandbox-denied hint");
2776
2777 assert!(hint.contains("read-only"), "{hint}");
2778 assert!(hint.contains("Ask-only escalation"), "{hint}");
2779 assert!(
2780 hint.contains("retry this exact command once with sandbox_permissions"),
2781 "{hint}"
2782 );
2783 assert!(hint.contains("justification"), "{hint}");
2784 }
2785
2786 #[test]
2787 fn contract_bash_denial_surfaces_the_escalation_shape() {
2788 let tmp = tempdir().expect("tempdir");
2789 let ctx =
2790 ToolContext::new(tmp.path()).with_elevated_sandbox_policy(ExecutionSandboxPolicy::ReadOnly);
2791 let mut result = failed_network_shell_result("", "Operation not permitted");
2792 result.sandbox_denied = true;
2793
2794 let error = finish_contract_bash_result(result, None, &ctx, None)
2795 .expect_err("sandbox denial is a failed call");
2796
2797 assert!(
2798 error
2799 .to_string()
2800 .contains("retry this exact command once with sandbox_permissions"),
2801 "{error}"
2802 );
2803 }
2804
2805 #[test]
2806 fn sandbox_denied_hint_absent_without_denial_or_policy() {
2807 let tmp = tempdir().expect("tempdir");
2808 let ctx =
2809 ToolContext::new(tmp.path()).with_elevated_sandbox_policy(ExecutionSandboxPolicy::ReadOnly);
2810 let undenied = failed_network_shell_result("", "No such file or directory");
2811 assert!(shell_sandbox_denied_hint(&ctx, &undenied).is_none());
2812
2813 let mut denied = failed_network_shell_result("", "");
2814 denied.sandbox_denied = true;
2815 let no_policy_ctx = ToolContext::new(tmp.path());
2816 assert!(shell_sandbox_denied_hint(&no_policy_ctx, &denied).is_none());
2817 }
2818
2819 #[test]
2820 fn shell_delta_result_surfaces_sandbox_denied_hint() {
2821 let tmp = tempdir().expect("tempdir");
2822 let ctx =
2823 ToolContext::new(tmp.path()).with_elevated_sandbox_policy(ExecutionSandboxPolicy::ReadOnly);
2824 let mut result = failed_network_shell_result("", "Operation not permitted");
2825 result.sandbox_denied = true;
2826
2827 let tool_result = build_shell_delta_tool_result(
2828 ShellDeltaResult {
2829 command: "touch out.txt".to_string(),
2830 result,
2831 stdout_total_len: 0,
2832 stderr_total_len: 0,
2833 },
2834 &ctx,
2835 );
2836
2837 assert!(
2838 tool_result
2839 .content
2840 .contains("The execution sandbox blocked this command"),
2841 "{}",
2842 tool_result.content
2843 );
2844 let metadata = tool_result.metadata.expect("metadata");
2845 assert!(metadata.get("sandbox_denied_hint").is_some());
2846 }
2847
2848 #[test]
2849 fn network_restricted_hint_ignores_local_failures() {
2850 let tmp = tempdir().expect("tempdir");
2851 let ctx = network_restricted_context(tmp.path());
2852 let result = failed_network_shell_result("", "No such file or directory");
2853
2854 assert!(shell_network_restricted_hint(&ctx, "cat missing.txt", &result).is_none());
2855 }
2856
2857 #[test]
2858 fn shell_delta_result_surfaces_network_restricted_hint() {
2859 let tmp = tempdir().expect("tempdir");
2860 let ctx = network_restricted_context(tmp.path());
2861 let result = failed_network_shell_result("000", "");
2862
2863 let tool_result = build_shell_delta_tool_result(
2864 ShellDeltaResult {
2865 command: "gh issue list".to_string(),
2866 result,
2867 stdout_total_len: 3,
2868 stderr_total_len: 0,
2869 },
2870 &ctx,
2871 );
2872
2873 assert!(!tool_result.success);
2874 assert!(tool_result.content.starts_with("Shell command blocked"));
2875 let metadata = tool_result.metadata.expect("metadata");
2876 assert_eq!(
2877 metadata
2878 .get("sandbox_network_restricted")
2879 .and_then(Value::as_bool),
2880 Some(true)
2881 );
2882 }
2883
2884 #[test]
2885 fn shell_delta_result_exposes_lossless_high_exit_code_and_hex() {
2886 let tmp = tempdir().expect("tempdir");
2887 let ctx = ToolContext::new(tmp.path());
2888 let mut result = failed_network_shell_result("", "");
2889 result.exit_code = Some(0xC000_0005);
2890
2891 let tool_result = build_shell_delta_tool_result(
2892 ShellDeltaResult {
2893 command: "echo probe".to_string(),
2894 result,
2895 stdout_total_len: 0,
2896 stderr_total_len: 0,
2897 },
2898 &ctx,
2899 );
2900
2901 assert!(
2902 tool_result
2903 .content
2904 .contains("exit code 3221225477 (0xC0000005)"),
2905 "{}",
2906 tool_result.content
2907 );
2908 let metadata = tool_result.metadata.expect("metadata");
2909 assert_eq!(metadata["exit_code"], json!(3221225477_i64));
2910 assert_eq!(metadata["exit_code_hex"], json!("0xC0000005"));
2911 }
2912
2913 #[test]
2914 fn shell_delta_result_surfaces_elapsed_time_in_content() {
2915 let tmp = tempdir().expect("tempdir");
2916 let ctx = ToolContext::new(tmp.path());
2917 let mut result = failed_network_shell_result("", "");
2918 result.status = ShellStatus::Running;
2919 result.duration_ms = 42_500;
2920 result.task_id = Some("shell-7".to_string());
2921
2922 let tool_result = build_shell_delta_tool_result(
2923 ShellDeltaResult {
2924 command: "cargo test --workspace".to_string(),
2925 result,
2926 stdout_total_len: 0,
2927 stderr_total_len: 0,
2928 },
2929 &ctx,
2930 );
2931
2932 assert!(
2933 tool_result
2934 .content
2935 .starts_with("Task shell-7 still running after 42.5 s."),
2936 "{}",
2937 tool_result.content
2938 );
2939 }
2940
2941 #[test]
2942 fn shell_delta_timing_line_omits_task_id_when_unknown() {
2943 let tmp = tempdir().expect("tempdir");
2944 let ctx = ToolContext::new(tmp.path());
2945 // failed_network_shell_result: ShellStatus::Failed, duration_ms: 25, task_id: None.
2946 let result = failed_network_shell_result("", "");
2947
2948 let tool_result = build_shell_delta_tool_result(
2949 ShellDeltaResult {
2950 command: "echo probe".to_string(),
2951 result,
2952 stdout_total_len: 0,
2953 stderr_total_len: 0,
2954 },
2955 &ctx,
2956 );
2957
2958 assert!(
2959 tool_result.content.starts_with("Task failed after 25 ms."),
2960 "{}",
2961 tool_result.content
2962 );
2963 }
2964
2965 #[test]
2966 fn shell_delta_timing_line_phrases_cover_terminal_statuses() {
2967 let tmp = tempdir().expect("tempdir");
2968 let ctx = ToolContext::new(tmp.path());
2969 for (status, phrase) in [
2970 (ShellStatus::Completed, "completed"),
2971 (ShellStatus::Killed, "killed"),
2972 (ShellStatus::TimedOut, "timed out"),
2973 ] {
2974 let mut result = failed_network_shell_result("", "");
2975 result.status = status;
2976 result.duration_ms = 5_000;
2977 let tool_result = build_shell_delta_tool_result(
2978 ShellDeltaResult {
2979 command: "echo probe".to_string(),
2980 result,
2981 stdout_total_len: 0,
2982 stderr_total_len: 0,
2983 },
2984 &ctx,
2985 );
2986 assert!(
2987 tool_result
2988 .content
2989 .starts_with(&format!("Task {phrase} after 5 s.")),
2990 "{}",
2991 tool_result.content
2992 );
2993 }
2994 }
2995
2996 #[test]
2997 fn shell_delta_timing_line_handles_zero_duration() {
2998 let tmp = tempdir().expect("tempdir");
2999 let ctx = ToolContext::new(tmp.path());
3000 let mut result = failed_network_shell_result("", "");
3001 result.status = ShellStatus::Completed;
3002 result.duration_ms = 0;
3003 let tool_result = build_shell_delta_tool_result(
3004 ShellDeltaResult {
3005 command: "echo probe".to_string(),
3006 result,
3007 stdout_total_len: 0,
3008 stderr_total_len: 0,
3009 },
3010 &ctx,
3011 );
3012 assert!(
3013 tool_result
3014 .content
3015 .starts_with("Task completed after 0 ms."),
3016 "{}",
3017 tool_result.content
3018 );
3019 }
3020
3021 #[test]
3022 fn shell_delta_timing_line_sits_below_network_hint() {
3023 let tmp = tempdir().expect("tempdir");
3024 let ctx = network_restricted_context(tmp.path());
3025 let result = failed_network_shell_result("000", "");
3026 let tool_result = build_shell_delta_tool_result(
3027 ShellDeltaResult {
3028 command: "gh issue list".to_string(),
3029 result,
3030 stdout_total_len: 3,
3031 stderr_total_len: 0,
3032 },
3033 &ctx,
3034 );
3035 let content = tool_result.content;
3036 let hint_pos = content.find("Shell command blocked").expect("hint present");
3037 let timing_pos = content
3038 .find("failed after 25 ms")
3039 .expect("timing line present");
3040 assert!(
3041 hint_pos < timing_pos,
3042 "hint must precede timing line: {content}"
3043 );
3044 }
3045
3046 #[test]
3047 fn shell_delta_result_includes_cargo_failure_summary() {
3048 let tmp = tempdir().expect("tempdir");
3049 let ctx = ToolContext::new(tmp.path());
3050 let result = ShellResult {
3051 task_id: None,
3052 status: ShellStatus::Failed,
3053 exit_code: Some(101),
3054 stdout: "running 1 test\ntest tests::fails ... FAILED\n\nfailures:\n\n---- tests::fails stdout ----\nthread 'tests::fails' panicked at src/lib.rs:7:9:\nboom\n\ntest result: FAILED. 0 passed; 1 failed; 0 ignored; finished in 0.00s\n".to_string(),
3055 stderr: "error: test failed, to rerun pass `--lib`".to_string(),
3056 duration_ms: 12,
3057 stdout_len: 0,
3058 stderr_len: 0,
3059 stdout_omitted: 0,
3060 stderr_omitted: 0,
3061 stdout_truncated: false,
3062 stderr_truncated: false,
3063 sandboxed: false,
3064 sandbox_type: None,
3065 sandbox_denied: false,
3066 };
3067
3068 let tool_result = build_shell_delta_tool_result(
3069 ShellDeltaResult {
3070 command: "cargo test".to_string(),
3071 result,
3072 stdout_total_len: 0,
3073 stderr_total_len: 0,
3074 },
3075 &ctx,
3076 );
3077
3078 let metadata = tool_result.metadata.expect("metadata");
3079 assert_eq!(
3080 metadata["cargo_failure_summary"]["kind"],
3081 json!("test_failure")
3082 );
3083 assert!(
3084 metadata["cargo_failure_summary"]["summary"]
3085 .as_str()
3086 .unwrap()
3087 .contains("Failing tests: tests::fails")
3088 );
3089 assert!(
3090 metadata["summary"]
3091 .as_str()
3092 .unwrap()
3093 .contains("error: test failed")
3094 );
3095 }
3096
3097 #[test]
3098 fn shell_delta_result_keeps_existing_summary_for_generic_cargo_failure() {
3099 let tmp = tempdir().expect("tempdir");
3100 let ctx = ToolContext::new(tmp.path());
3101 let result = ShellResult {
3102 task_id: None,
3103 status: ShellStatus::Failed,
3104 exit_code: Some(1),
3105 stdout: "build failed".to_string(),
3106 stderr: "command failed without structured cargo diagnostics".to_string(),
3107 duration_ms: 12,
3108 stdout_len: 0,
3109 stderr_len: 0,
3110 stdout_omitted: 0,
3111 stderr_omitted: 0,
3112 stdout_truncated: false,
3113 stderr_truncated: false,
3114 sandboxed: false,
3115 sandbox_type: None,
3116 sandbox_denied: false,
3117 };
3118
3119 let tool_result = build_shell_delta_tool_result(
3120 ShellDeltaResult {
3121 command: "cargo test".to_string(),
3122 result,
3123 stdout_total_len: 0,
3124 stderr_total_len: 0,
3125 },
3126 &ctx,
3127 );
3128
3129 let metadata = tool_result.metadata.expect("metadata");
3130 assert!(metadata.get("cargo_failure_summary").is_none());
3131 assert_eq!(
3132 metadata["summary"],
3133 json!("command failed without structured cargo diagnostics")
3134 );
3135 }
3136
3137 #[test]
3138 fn shell_delta_result_surfaces_python_build_dependency_hint() {
3139 let tmp = tempdir().expect("tempdir");
3140 let ctx = ToolContext::new(tmp.path());
3141 let result = ShellResult {
3142 task_id: None,
3143 status: ShellStatus::Failed,
3144 exit_code: Some(1),
3145 stdout: String::new(),
3146 stderr: "running build_ext\nModuleNotFoundError: No module named 'setuptools'\n"
3147 .to_string(),
3148 duration_ms: 12,
3149 stdout_len: 0,
3150 stderr_len: 72,
3151 stdout_omitted: 0,
3152 stderr_omitted: 0,
3153 stdout_truncated: false,
3154 stderr_truncated: false,
3155 sandboxed: false,
3156 sandbox_type: None,
3157 sandbox_denied: false,
3158 };
3159
3160 let tool_result = build_shell_delta_tool_result(
3161 ShellDeltaResult {
3162 command: "python setup.py build_ext --inplace".to_string(),
3163 result,
3164 stdout_total_len: 0,
3165 stderr_total_len: 72,
3166 },
3167 &ctx,
3168 );
3169
3170 assert!(!tool_result.success);
3171 assert!(
3172 tool_result
3173 .content
3174 .starts_with("Python build dependency missing")
3175 );
3176 let metadata = tool_result.metadata.expect("metadata");
3177 assert_eq!(
3178 metadata["python_build_dependency_hint"]["kind"],
3179 json!("missing_setuptools")
3180 );
3181 assert!(
3182 metadata["python_build_dependency_hint"]["hint"]
3183 .as_str()
3184 .unwrap()
3185 .contains("setuptools")
3186 );
3187 }
3188
3189 #[test]
3190 fn test_summarize_output_strips_truncation_note() {
3191 let long_output = "x".repeat(60_000);
3192 let (truncated, _meta) = truncate_with_meta(&long_output);
3193 let summary = summarize_output(&truncated);
3194 assert!(!summary.contains("Output truncated at"));
3195 }
3196
3197 #[tokio::test]
3198 async fn test_exec_shell_metadata_includes_summaries() {
3199 let tmp = tempdir().expect("tempdir");
3200 let ctx = ToolContext::new(tmp.path());
3201 let tool = BashTool::new("Bash");
3202
3203 let result = tool
3204 .execute(json!({"command": echo_command("hello")}), &ctx)
3205 .await
3206 .expect("execute");
3207 assert!(result.success);
3208
3209 let meta = result.metadata.expect("metadata");
3210 let summary = meta
3211 .get("summary")
3212 .and_then(Value::as_str)
3213 .unwrap_or_default()
3214 .to_string();
3215 assert!(summary.contains("hello"));
3216 assert!(meta.get("stdout_len").is_some());
3217 assert!(meta.get("stdout_truncated").is_some());
3218 }
3219
3220 #[cfg(not(windows))]
3221 #[tokio::test]
3222 async fn test_exec_shell_combined_output_uses_single_stream() {
3223 let tmp = tempdir().expect("tempdir");
3224 let ctx = ToolContext::new(tmp.path());
3225 let tool = BashTool::new("Bash");
3226 let command = "printf 'out\\n'; printf 'err\\n' >&2";
3227
3228 let result = tool
3229 .execute(json!({"command": command, "combined_output": true}), &ctx)
3230 .await
3231 .expect("execute");
3232 assert!(result.success, "{}", result.content);
3233 assert!(result.content.contains("out"), "{}", result.content);
3234 assert!(result.content.contains("err"), "{}", result.content);
3235
3236 let meta = result.metadata.expect("metadata");
3237 assert_eq!(
3238 meta.get("combined_output").and_then(Value::as_bool),
3239 Some(true)
3240 );
3241 }
3242
3243 #[tokio::test]
3244 async fn test_exec_shell_foreground_timeout_guides_background_rerun() {
3245 let tmp = tempdir().expect("tempdir");
3246 let ctx = ToolContext::new(tmp.path());
3247 let tool = BashTool::new("Bash");
3248
3249 let result = tool
3250 .execute(
3251 json!({
3252 "command": sleep_command(10),
3253 "timeout_ms": 1000
3254 }),
3255 &ctx,
3256 )
3257 .await
3258 .expect("execute");
3259
3260 assert!(!result.success);
3261 // The rerun instruction has to be spelled in the canonical action form:
3262 // `exec_shell` / `task_shell_start` are not both dispatchable, and the
3263 // model can only reach the shell through `Bash`.
3264 assert!(
3265 result
3266 .content
3267 .contains("Bash action=\"run\" background=true")
3268 );
3269 assert!(result.content.contains("Bash action=\"wait\""));
3270 assert!(!result.content.contains("exec_shell"));
3271 assert!(result.content.contains("process killed"));
3272 let meta = result.metadata.expect("metadata");
3273 assert_eq!(meta.get("status").and_then(Value::as_str), Some("TimedOut"));
3274 let recovery = meta
3275 .get("foreground_timeout_recovery")
3276 .expect("timeout recovery metadata");
3277 assert_eq!(
3278 recovery
3279 .get("rerun_as")
3280 .and_then(|rerun| rerun.get("background"))
3281 .and_then(Value::as_bool),
3282 Some(true)
3283 );
3284 assert_eq!(
3285 recovery
3286 .get("rerun_as")
3287 .and_then(|rerun| rerun.get("tool"))
3288 .and_then(Value::as_str),
3289 Some("Bash")
3290 );
3291 let hint = recovery
3292 .get("hint")
3293 .and_then(Value::as_str)
3294 .unwrap_or_default();
3295 assert!(hint.contains("Bash action=\"wait\""), "{hint}");
3296 assert!(!hint.contains("exec_shell"), "{hint}");
3297 // The structured tool list is read by the model too; it must not hand
3298 // over names the registry does not resolve.
3299 let recommended = recovery.to_string();
3300 assert!(!recommended.contains("exec_shell"), "{recommended}");
3301 }
3302
3303 #[test]
3304 fn background_schema_distinguishes_temporary_jobs_from_persistent_services() {
3305 let schema = BashTool::new("Bash").input_schema();
3306 let d = schema["properties"]["background"]["description"]
3307 .as_str()
3308 .expect("background description");
3309 assert!(d.contains("killed") && d.contains("background:true") && d.contains("persist:true"));
3310 }
3311
3312 #[tokio::test]
3313 async fn test_exec_shell_foreground_cancel_kills_process() {
3314 let tmp = tempdir().expect("tempdir");
3315 let cancel_token = tokio_util::sync::CancellationToken::new();
3316 let ctx = ToolContext::new(tmp.path()).with_cancel_token(cancel_token.clone());
3317 let command = sleep_command(30);
3318
3319 let task = tokio::spawn(async move {
3320 BashTool::new("Bash")
3321 .execute(
3322 json!({
3323 "command": command,
3324 "timeout_ms": 600_000
3325 }),
3326 &ctx,
3327 )
3328 .await
3329 .expect("execute")
3330 });
3331
3332 tokio::time::sleep(Duration::from_millis(150)).await;
3333 cancel_token.cancel();
3334
3335 let result = tokio::time::timeout(Duration::from_secs(5), task)
3336 .await
3337 .expect("foreground shell should observe cancellation")
3338 .expect("task should not panic");
3339
3340 assert!(!result.success);
3341 assert!(result.content.contains("Command canceled"));
3342 let meta = result.metadata.expect("metadata");
3343 assert_eq!(meta.get("status").and_then(Value::as_str), Some("Killed"));
3344 assert_eq!(meta.get("canceled").and_then(Value::as_bool), Some(true));
3345 }
3346
3347 #[tokio::test]
3348 async fn test_exec_shell_foreground_can_move_to_background() {
3349 let tmp = tempdir().expect("tempdir");
3350 let ctx = ToolContext::new(tmp.path());
3351 let shell_manager = ctx.shell_manager.clone();
3352 let command = sleep_command(30);
3353 let task_ctx = ctx.clone();
3354
3355 let task = tokio::spawn(async move {
3356 BashTool::new("Bash")
3357 .execute(
3358 json!({
3359 "command": command,
3360 "timeout_ms": 600_000
3361 }),
3362 &task_ctx,
3363 )
3364 .await
3365 .expect("execute")
3366 });
3367
3368 tokio::time::sleep(Duration::from_millis(150)).await;
3369 shell_manager
3370 .lock()
3371 .expect("shell manager lock")
3372 .request_foreground_background();
3373
3374 let result = tokio::time::timeout(Duration::from_secs(5), task)
3375 .await
3376 .expect("foreground shell should detach")
3377 .expect("task should not panic");
3378
3379 assert!(result.success);
3380 assert!(
3381 result
3382 .content
3383 .contains("Foreground shell wait moved to /jobs")
3384 );
3385 // The detach message points the model at the wait action for early
3386 // output, and hands over the task_id it needs to make that call.
3387 assert!(
3388 result.content.contains("Bash action=\"wait\""),
3389 "{}",
3390 result.content
3391 );
3392 assert!(result.content.contains("task_id="), "{}", result.content);
3393 assert!(!result.content.contains("exec_shell"), "{}", result.content);
3394
3395 let meta = result.metadata.expect("metadata");
3396 assert_eq!(meta.get("status").and_then(Value::as_str), Some("Running"));
3397 assert_eq!(
3398 meta.get("backgrounded").and_then(Value::as_bool),
3399 Some(true)
3400 );
3401 let task_id = meta
3402 .get("task_id")
3403 .and_then(Value::as_str)
3404 .expect("task id")
3405 .to_string();
3406
3407 let mut manager = shell_manager.lock().expect("shell manager lock");
3408 let job = manager.inspect_job(&task_id).expect("inspect job");
3409 assert_eq!(job.snapshot.status, ShellStatus::Running);
3410 assert!(
3411 job.snapshot.background,
3412 "detached foreground work needs a background receipt"
3413 );
3414 let killed = manager.kill(&task_id).expect("kill");
3415 assert_eq!(killed.status, ShellStatus::Killed);
3416 }
3417
3418 #[cfg(unix)]
3419 #[tokio::test]
3420 async fn dropped_foreground_wait_kills_descendants_and_retains_unreceived_output() {
3421 let tmp = tempdir().unwrap();
3422 let pid_file = tmp.path().join("descendant.pid");
3423 let command = format!(
3424 "printf 'retained-before-drop\\n'; printf '%{}s\\n' x; \
3425 CODEWHALE_SHELL_DESCENDANT_HELPER=1 \
3426 CODEWHALE_SHELL_DESCENDANT_PID_FILE={} {} --exact \
3427 tools::shell::tests::shell_descendant_helper_process --nocapture",
3428 RAW_STREAM_SETTLED_TAIL_BYTES + 1024,
3429 shell_words::quote(&pid_file.display().to_string()),
3430 shell_words::quote(&std::env::current_exe().unwrap().display().to_string()),
3431 );
3432 let ctx = ToolContext::new(tmp.path()).with_state_namespace("foreground-drop".to_string());
3433 let manager = ctx.shell_manager.clone();
3434 let task_ctx = ctx.clone();
3435 let task = tokio::spawn(async move {
3436 BashTool::new("Bash")
3437 .execute(
3438 json!({"command": command, "timeout_ms": 600_000}),
3439 &task_ctx,
3440 )
3441 .await
3442 });
3443 let descendant = tokio::time::timeout(Duration::from_secs(10), async {
3444 loop {
3445 if let Ok(raw) = std::fs::read_to_string(&pid_file)
3446 && let Ok(pid) = raw.trim().parse::<libc::pid_t>()
3447 {
3448 break pid;
3449 }
3450 tokio::time::sleep(Duration::from_millis(10)).await;
3451 }
3452 })
3453 .await
3454 .expect("descendant must start");
3455 task.abort();
3456 assert!(task.await.unwrap_err().is_cancelled());
3457 assert!(
3458 wait_for_shell_pid_exit(descendant),
3459 "dropping the wait must stop its descendant"
3460 );
3461 let mut manager = manager.lock().unwrap();
3462 let jobs = manager.list_jobs_for_session("foreground-drop");
3463 assert_eq!(jobs.len(), 1);
3464 assert_eq!(jobs[0].status, ShellStatus::Killed);
3465 let detail = manager
3466 .inspect_job(&jobs[0].id)
3467 .expect("inspect cancelled job");
3468 assert!(detail.stdout.contains("retained-before-drop"));
3469 assert!(detail.stdout.len() > RAW_STREAM_SETTLED_TAIL_BYTES);
3470 assert!(!manager.has_finished_unreported_jobs_for_session("foreground-drop"));
3471 }
3472
3473 #[tokio::test]
3474 async fn lowercase_bash_foreground_detach_is_a_successful_running_receipt() {
3475 let tmp = tempdir().expect("tempdir");
3476 let ctx = ToolContext::new(tmp.path());
3477 let shell_manager = ctx.shell_manager.clone();
3478 let command = sleep_command(30);
3479 let task_ctx = ctx.clone();
3480
3481 let task = tokio::spawn(async move {
3482 LowercaseBashTool
3483 .execute(json!({"command": command}), &task_ctx)
3484 .await
3485 .expect("execute")
3486 });
3487
3488 tokio::time::sleep(Duration::from_millis(150)).await;
3489 shell_manager
3490 .lock()
3491 .expect("shell manager lock")
3492 .request_foreground_background();
3493
3494 let result = tokio::time::timeout(Duration::from_secs(5), task)
3495 .await
3496 .expect("foreground shell should detach")
3497 .expect("task should not panic");
3498
3499 assert!(result.success, "{}", result.content);
3500 assert!(
3501 result.content.contains("moved to /jobs"),
3502 "{}",
3503 result.content
3504 );
3505 assert!(!result.content.contains("code -1"), "{}", result.content);
3506 let metadata = result.metadata.expect("metadata");
3507 assert_eq!(metadata["status"], "Running");
3508 assert_eq!(metadata["backgrounded"], true);
3509 let task_id = metadata["task_id"].as_str().expect("task id");
3510
3511 let mut manager = shell_manager.lock().expect("shell manager lock");
3512 let job = manager.inspect_job(task_id).expect("inspect job");
3513 assert_eq!(job.snapshot.status, ShellStatus::Running);
3514 manager.kill(task_id).expect("kill test job");
3515 }
3516
3517 #[tokio::test]
3518 async fn test_exec_shell_wait_cancel_leaves_background_process_running() {
3519 let tmp = tempdir().expect("tempdir");
3520 let cancel_token = tokio_util::sync::CancellationToken::new();
3521 let ctx = ToolContext::new(tmp.path()).with_cancel_token(cancel_token.clone());
3522 let shell_manager = ctx.shell_manager.clone();
3523 let started = execute_shell(
3524 &mut shell_manager.lock().expect("shell manager lock"),
3525 &sleep_command(30),
3526 None,
3527 600_000,
3528 true,
3529 )
3530 .expect("execute");
3531 let task_id = started.task_id.expect("task id");
3532 let wait_task_id = task_id.clone();
3533 let task_ctx = ctx.clone();
3534
3535 let task = tokio::spawn(async move {
3536 BashTool::new("Bash")
3537 .execute(
3538 json!({
3539 "action": "wait",
3540 "task_id": wait_task_id,
3541 "timeout_ms": 600_000
3542 }),
3543 &task_ctx,
3544 )
3545 .await
3546 .expect("wait")
3547 });
3548
3549 tokio::time::sleep(Duration::from_millis(150)).await;
3550 cancel_token.cancel();
3551
3552 let result = tokio::time::timeout(Duration::from_secs(5), task)
3553 .await
3554 .expect("wait should observe cancellation")
3555 .expect("task should not panic");
3556
3557 assert!(result.success);
3558 assert!(result.content.contains("still running"));
3559 let meta = result.metadata.expect("metadata");
3560 assert_eq!(meta.get("status").and_then(Value::as_str), Some("Running"));
3561 assert_eq!(
3562 meta.get("wait_canceled").and_then(Value::as_bool),
3563 Some(true)
3564 );
3565
3566 let mut manager = shell_manager.lock().expect("shell manager lock");
3567 let job = manager.inspect_job(&task_id).expect("inspect job");
3568 assert_eq!(job.snapshot.status, ShellStatus::Running);
3569 let killed = manager.kill(&task_id).expect("kill");
3570 assert_eq!(killed.status, ShellStatus::Killed);
3571 }
3572
3573 #[tokio::test]
3574 async fn test_completed_background_shell_releases_process_handles() {
3575 let tmp = tempdir().expect("tempdir");
3576 let ctx = ToolContext::new(tmp.path());
3577 let shell_manager = ctx.shell_manager.clone();
3578 let started = execute_shell(
3579 &mut shell_manager.lock().expect("shell manager lock"),
3580 &echo_command("done"),
3581 None,
3582 600_000,
3583 true,
3584 )
3585 .expect("execute");
3586 let task_id = started.task_id.expect("task id");
3587
3588 let result = BashTool::alias("exec_shell_wait", "wait")
3589 .execute(
3590 json!({
3591 "task_id": task_id.clone(),
3592 "wait": true,
3593 "timeout_ms": BACKGROUND_COMPLETION_WAIT_MS
3594 }),
3595 &ctx,
3596 )
3597 .await
3598 .expect("wait");
3599
3600 assert!(result.success);
3601 let mut manager = shell_manager.lock().expect("shell manager lock");
3602 let result = wait_for_completed_shell(&mut manager, &task_id);
3603 assert_eq!(result.status, ShellStatus::Completed);
3604 let shell = manager.processes.get_mut(&task_id).expect("tracked shell");
3605 shell.poll();
3606 assert_eq!(shell.status, ShellStatus::Completed);
3607 assert!(shell.stdin.is_none());
3608 assert!(shell.child.is_none());
3609 assert!(shell.stdout_thread.is_none());
3610 assert!(shell.stderr_thread.is_none());
3611 }
3612
3613 #[cfg(unix)]
3614 #[tokio::test]
3615 async fn exec_shell_cancel_kills_descendant_process_group() {
3616 let tmp = tempdir().expect("tempdir");
3617 let pid_file = tmp.path().join("descendant.pid");
3618 let test_binary = std::env::current_exe().expect("current test binary");
3619 let command = format!(
3620 "{} --exact {} --nocapture",
3621 shell_words::quote(&test_binary.display().to_string()),
3622 shell_words::quote("tools::shell::tests::shell_descendant_helper_process"),
3623 );
3624 let ctx = ToolContext::new(tmp.path());
3625 let mut env = std::collections::HashMap::new();
3626 env.insert(SHELL_DESCENDANT_HELPER_ENV.to_string(), "1".to_string());
3627 env.insert(
3628 SHELL_DESCENDANT_PID_FILE_ENV.to_string(),
3629 pid_file.display().to_string(),
3630 );
3631 let started = ctx
3632 .shell_manager
3633 .lock()
3634 .expect("shell manager")
3635 .execute_with_options_env_for_session(
3636 &command,
3637 None,
3638 60_000,
3639 true,
3640 None,
3641 false,
3642 None,
3643 env,
3644 &ctx.state_namespace,
3645 )
3646 .expect("start descendant tree");
3647 let task_id = started.task_id.expect("task id");
3648 let descendant = wait_for_shell_pid_file(&pid_file);
3649
3650 let result = BashTool::alias("exec_shell_cancel", "cancel")
3651 .execute(json!({"task_id": task_id}), &ctx)
3652 .await
3653 .expect("cancel process group");
3654 assert!(result.success);
3655 assert!(
3656 wait_for_shell_pid_exit(descendant),
3657 "descendant {descendant} survived shell process-group cancellation"
3658 );
3659 }
3660
3661 #[tokio::test]
3662 async fn test_exec_shell_cancel_tool_kills_background_process() {
3663 let tmp = tempdir().expect("tempdir");
3664 let ctx = ToolContext::new(tmp.path());
3665 let shell_manager = ctx.shell_manager.clone();
3666 let started = execute_shell(
3667 &mut shell_manager.lock().expect("shell manager lock"),
3668 &sleep_command(30),
3669 None,
3670 600_000,
3671 true,
3672 )
3673 .expect("execute");
3674 let task_id = started.task_id.expect("task id");
3675
3676 let result = BashTool::alias("exec_shell_cancel", "cancel")
3677 .execute(json!({ "task_id": task_id }), &ctx)
3678 .await
3679 .expect("cancel");
3680
3681 assert!(result.success);
3682 assert!(result.content.contains("Canceled background command"));
3683 let meta = result.metadata.expect("metadata");
3684 assert_eq!(meta.get("status").and_then(Value::as_str), Some("Killed"));
3685
3686 let task_id = meta
3687 .get("task_id")
3688 .and_then(Value::as_str)
3689 .expect("task id");
3690 let mut manager = shell_manager.lock().expect("shell manager lock");
3691 let job = manager.inspect_job(task_id).expect("inspect job");
3692 assert_eq!(job.snapshot.status, ShellStatus::Killed);
3693 }
3694
3695 #[tokio::test]
3696 async fn test_exec_shell_cancel_tool_can_kill_all_running_processes() {
3697 let tmp = tempdir().expect("tempdir");
3698 let ctx = ToolContext::new(tmp.path());
3699 let shell_manager = ctx.shell_manager.clone();
3700 let first = execute_shell(
3701 &mut shell_manager.lock().expect("shell manager lock"),
3702 &sleep_command(30),
3703 None,
3704 600_000,
3705 true,
3706 )
3707 .expect("execute first")
3708 .task_id
3709 .expect("first task id");
3710 let second = execute_shell(
3711 &mut shell_manager.lock().expect("shell manager lock"),
3712 &sleep_command(30),
3713 None,
3714 600_000,
3715 true,
3716 )
3717 .expect("execute second")
3718 .task_id
3719 .expect("second task id");
3720
3721 let result = BashTool::alias("exec_shell_cancel", "cancel")
3722 .execute(json!({ "all": true }), &ctx)
3723 .await
3724 .expect("cancel all");
3725
3726 assert!(result.success);
3727 let meta = result.metadata.expect("metadata");
3728 assert_eq!(meta.get("status").and_then(Value::as_str), Some("Killed"));
3729 assert_eq!(meta.get("canceled").and_then(Value::as_u64), Some(2));
3730
3731 let mut manager = shell_manager.lock().expect("shell manager lock");
3732 let first_job = manager.inspect_job(&first).expect("inspect first");
3733 let second_job = manager.inspect_job(&second).expect("inspect second");
3734 assert_eq!(first_job.snapshot.status, ShellStatus::Killed);
3735 assert_eq!(second_job.snapshot.status, ShellStatus::Killed);
3736 }
3737
3738 fn make_failed_result(stderr: &str) -> ShellResult {
3739 ShellResult {
3740 task_id: None,
3741 status: ShellStatus::Failed,
3742 exit_code: Some(1),
3743 stdout: String::new(),
3744 stderr: stderr.to_string(),
3745 duration_ms: 0,
3746 stdout_len: 0,
3747 stderr_len: stderr.len(),
3748 stdout_omitted: 0,
3749 stderr_omitted: 0,
3750 stdout_truncated: false,
3751 sandboxed: false,
3752 sandbox_type: None,
3753 sandbox_denied: false,
3754 stderr_truncated: false,
3755 }
3756 }
3757
3758 #[test]
3759 fn test_macos_provenance_detected_by_activity_time_message() {
3760 let result = make_failed_result(
3761 "failed to update builder last activity time: open \
3762 /Users/user/.docker/buildx/activity/.tmp-abc: operation not permitted",
3763 );
3764 assert!(looks_like_macos_provenance_failure(&result));
3765 }
3766
3767 #[test]
3768 fn test_macos_provenance_detected_by_activity_path_and_eperm() {
3769 let result = make_failed_result(
3770 "error: open /home/user/.docker/buildx/activity/foo: operation not permitted",
3771 );
3772 assert!(looks_like_macos_provenance_failure(&result));
3773 }
3774
3775 #[test]
3776 fn test_macos_provenance_not_triggered_on_success() {
3777 let mut result = make_failed_result(
3778 "failed to update builder last activity time: open \
3779 /Users/user/.docker/buildx/activity/.tmp-abc: operation not permitted",
3780 );
3781 result.status = ShellStatus::Completed;
3782 result.exit_code = Some(0);
3783 assert!(!looks_like_macos_provenance_failure(&result));
3784 }
3785
3786 #[test]
3787 fn test_macos_provenance_not_triggered_on_unrelated_eperm() {
3788 let result = make_failed_result("open /some/other/path: operation not permitted");
3789 assert!(!looks_like_macos_provenance_failure(&result));
3790 }
3791
3792 // Regression test for #828: shell spawns an orphaned background subprocess
3793 // (simulating `nohup curl`) that keeps the pipe write-end open after the shell
3794 // exits. collect_output() must not block indefinitely — it kills the whole
3795 // process group first, allowing reader threads to get EOF and exit.
3796 #[cfg(unix)]
3797 #[test]
3798 fn test_orphaned_subprocess_does_not_block_collect_output() {
3799 let tmp = tempdir().expect("tempdir");
3800 let mut manager = ShellManager::new(tmp.path().to_path_buf());
3801
3802 // sh spawns `sleep 100 &` and exits; the sleep subprocess inherits the
3803 // pipe write-ends and would keep reader threads blocked without the fix.
3804 let result =
3805 execute_shell(&mut manager, "sh -c 'sleep 100 &'", None, 5000, true).expect("execute");
3806 let task_id = result.task_id.expect("task id");
3807
3808 // Drive to completion with a tight timeout — must not hang.
3809 let done = manager
3810 .get_output(&task_id, true, 3000)
3811 .expect("get_output must complete, not hang");
3812 assert_eq!(done.status, ShellStatus::Completed);
3813 }
3814
3815 #[cfg(unix)]
3816 #[test]
3817 fn foreground_shell_does_not_block_on_orphaned_subprocess_pipe() {
3818 let tmp = tempdir().expect("tempdir");
3819 let mut manager = ShellManager::new(tmp.path().to_path_buf());
3820
3821 let started = std::time::Instant::now();
3822 let result = execute_shell(&mut manager, "sh -c 'sleep 100 &'", None, 5000, false)
3823 .expect("foreground execute must complete, not hang");
3824
3825 assert!(
3826 started.elapsed() < std::time::Duration::from_secs(4),
3827 "foreground execute blocked on descendant pipe handles"
3828 );
3829 assert_eq!(result.status, ShellStatus::Completed);
3830 }
3831
3832 // Windows equivalent of the orphaned pipe-handle regression. `cmd /c start /b`
3833 // launches a descendant process that inherits stdout/stderr and outlives the
3834 // shell. Job-object cleanup must terminate that descendant before reader-thread
3835 // joins, otherwise get_output() blocks until ping exits.
3836 #[cfg(windows)]
3837 #[test]
3838 fn background_collection_does_not_block_on_detached_descendant_pipe() {
3839 let tmp = tempdir().expect("tempdir");
3840 let mut manager = ShellManager::new(tmp.path().to_path_buf());
3841
3842 let result = execute_shell(
3843 &mut manager,
3844 r#"cmd /c start "" /b ping 127.0.0.1 -n 4"#,
3845 None,
3846 5000,
3847 true,
3848 )
3849 .expect("execute");
3850 let task_id = result.task_id.expect("task id");
3851
3852 let started = std::time::Instant::now();
3853 let done = manager
3854 .get_output(&task_id, true, 3000)
3855 .expect("get_output must complete, not hang");
3856
3857 assert!(
3858 started.elapsed() < std::time::Duration::from_secs(6),
3859 "get_output blocked on descendant pipe handles"
3860 );
3861 assert_eq!(done.status, ShellStatus::Completed);
3862 }
3863
3864 #[cfg(windows)]
3865 #[test]
3866 fn windows_job_terminate_denied_falls_back_to_child_kill() {
3867 let mut child = Command::new("ping")
3868 .args(["127.0.0.1", "-n", "20"])
3869 .stdin(Stdio::null())
3870 .stdout(Stdio::null())
3871 .stderr(Stdio::null())
3872 .spawn()
3873 .expect("spawn ping");
3874
3875 let job = WindowsJob::attach_to_child(&child).expect("attach job");
3876 let limited_job = duplicate_job_without_terminate_access(job);
3877
3878 assert!(
3879 limited_job.terminate().is_err(),
3880 "limited job handle should not allow TerminateJobObject"
3881 );
3882
3883 terminate_child_and_close_windows_job(Some(limited_job), &mut child)
3884 .expect("fallback child kill");
3885
3886 let status = child
3887 .wait_timeout(std::time::Duration::from_secs(3))
3888 .expect("wait after fallback kill");
3889 assert!(
3890 status.is_some(),
3891 "fallback child kill should terminate child"
3892 );
3893 }
3894
3895 #[cfg(windows)]
3896 #[test]
3897 fn windows_job_close_releases_foreground_reader_threads_when_terminate_denied() {
3898 let mut child = Command::new("ping")
3899 .args(["127.0.0.1", "-n", "8"])
3900 .stdin(Stdio::null())
3901 .stdout(Stdio::piped())
3902 .stderr(Stdio::piped())
3903 .spawn()
3904 .expect("spawn ping");
3905
3906 let job = WindowsJob::attach_to_child(&child).expect("attach job");
3907 let limited_job = duplicate_job_without_terminate_access(job);
3908 assert!(
3909 limited_job.terminate().is_err(),
3910 "limited job handle should not allow TerminateJobObject"
3911 );
3912
3913 let stdout_handle = child.stdout.take().expect("stdout pipe");
3914 let stderr_handle = child.stderr.take().expect("stderr pipe");
3915 let stdout_thread = std::thread::spawn(move || {
3916 let mut reader = stdout_handle;
3917 let mut buf = Vec::new();
3918 let _ = reader.read_to_end(&mut buf);
3919 buf
3920 });
3921 let stderr_thread = std::thread::spawn(move || {
3922 let mut reader = stderr_handle;
3923 let mut buf = Vec::new();
3924 let _ = reader.read_to_end(&mut buf);
3925 buf
3926 });
3927
3928 let started = std::time::Instant::now();
3929 terminate_and_close_windows_job(Some(limited_job));
3930 let _ = stdout_thread.join().unwrap_or_default();
3931 let _ = stderr_thread.join().unwrap_or_default();
3932 let status = child
3933 .wait_timeout(std::time::Duration::from_secs(3))
3934 .expect("wait after kill-on-close");
3935
3936 assert!(
3937 started.elapsed() < std::time::Duration::from_secs(4),
3938 "reader joins waited for natural descendant exit instead of kill-on-close"
3939 );
3940 assert!(status.is_some(), "kill-on-close should terminate child");
3941 }
3942
3943 #[cfg(windows)]
3944 #[test]
3945 fn windows_job_kill_on_close_releases_reader_threads_when_terminate_denied() {
3946 let tmp = tempdir().expect("tempdir");
3947 let mut manager = ShellManager::new(tmp.path().to_path_buf());
3948
3949 let result = execute_shell(
3950 &mut manager,
3951 r#"cmd /c start "" /b ping 127.0.0.1 -n 8"#,
3952 None,
3953 5000,
3954 true,
3955 )
3956 .expect("execute");
3957 let task_id = result.task_id.expect("task id");
3958
3959 {
3960 let shell = manager
3961 .processes
3962 .get_mut(&task_id)
3963 .expect("background shell");
3964 let job = shell.windows_job.take().expect("windows job attached");
3965 let limited_job = duplicate_job_without_terminate_access(job);
3966 assert!(
3967 limited_job.terminate().is_err(),
3968 "limited job handle should not allow TerminateJobObject"
3969 );
3970 shell.windows_job = Some(limited_job);
3971 }
3972
3973 let started = std::time::Instant::now();
3974 let done = manager
3975 .get_output(&task_id, true, 3000)
3976 .expect("get_output must complete via kill-on-close fallback");
3977
3978 assert!(
3979 started.elapsed() < std::time::Duration::from_secs(4),
3980 "get_output waited for natural descendant exit instead of kill-on-close"
3981 );
3982 assert_eq!(done.status, ShellStatus::Completed);
3983 }
3984
3985 #[cfg(windows)]
3986 #[test]
3987 fn killed_shell_does_not_wait_for_blocked_reader_threads() {
3988 let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
3989 let stdout_thread = std::thread::spawn(move || {
3990 let _ = release_rx.recv();
3991 });
3992 let now = std::time::Instant::now();
3993 let mut shell = BackgroundShell {
3994 id: "killed-reader".to_string(),
3995 owner_session_id: "windows-test-session".to_string(),
3996 command: "test".to_string(),
3997 working_dir: std::path::PathBuf::from("."),
3998 status: ShellStatus::Killed,
3999 exit_code: None,
4000 started_at: now,
4001 finished_at: Some(now),
4002 finished_at_utc: Some(chrono::Utc::now()),
4003 background: true,
4004 last_output_at: now,
4005 last_observed_output_len: 0,
4006 sandbox_type: SandboxType::None,
4007 ownership: ShellOwnership::Managed,
4008 linked_task_id: None,
4009 owner_agent: None,
4010 origin_tool_call_id: None,
4011 origin_turn_id: None,
4012 stdout_buffer: super::new_shared_raw_output(),
4013 stderr_buffer: None,
4014 heavy_permit: None,
4015 stdout_cursor: 0,
4016 stderr_cursor: 0,
4017 completion_reported: false,
4018 bounded_output: None,
4019 stdin: None,
4020 pty_master: None,
4021 terminal_size: None,
4022 child: None,
4023 windows_job: None,
4024 stdout_thread: Some(stdout_thread),
4025 stderr_thread: None,
4026 work_lifecycle: None,
4027 lifecycle_seq: 0,
4028 last_lifecycle_status: None,
4029 last_lifecycle_bytes: 0,
4030 wait_failed: false,
4031 };
4032
4033 let started = std::time::Instant::now();
4034 shell.collect_output();
4035
4036 assert!(
4037 started.elapsed() < std::time::Duration::from_secs(1),
4038 "killed shell must not synchronously join a blocked reader"
4039 );
4040 release_tx.send(()).expect("release detached reader");
4041 }
4042
4043 #[test]
4044 fn test_list_jobs_cleans_up_completed_old_processes() {
4045 let tmp = tempdir().expect("tempdir");
4046 let mut manager = ShellManager::new(tmp.path().to_path_buf());
4047
4048 let bg =
4049 execute_shell(&mut manager, &echo_command("bg"), None, 5000, true).expect("execute bg");
4050 let bg_id = bg.task_id.expect("bg task id");
4051 manager.get_output(&bg_id, true, 3000).expect("bg done");
4052
4053 // Both the completed job and any tracking state should be present.
4054 assert!(!manager.processes.is_empty());
4055
4056 // cleanup(ZERO) removes every delivered completed process immediately.
4057 manager.drain_finished_jobs_with_evidence();
4058 manager.cleanup(Duration::ZERO);
4059 assert!(
4060 manager.processes.is_empty(),
4061 "completed processes should be evicted by cleanup"
4062 );
4063 }
4064
4065 /// A job that ran longer than the retention age must survive until its
4066 /// completion is delivered: age counts from the finish, not the start.
4067 #[test]
4068 fn cleanup_ages_jobs_from_finish() {
4069 let tmp = tempdir().expect("tempdir");
4070 let mut manager = ShellManager::new(tmp.path().to_path_buf());
4071 let long_run = FINISHED_SHELL_MAX_AGE + Duration::from_secs(600);
4072 manager.seed_finished_record_for_test("long-job", long_run);
4073
4074 // Started 70 minutes ago, finished just now, not yet delivered.
4075 manager.list_jobs();
4076 assert!(
4077 manager.inspect_job("long-job").is_ok(),
4078 "a long job must not be evicted the moment it finishes"
4079 );
4080 let delivered = manager.drain_finished_jobs_with_evidence();
4081 assert_eq!(delivered.len(), 1, "the completion must still be delivered");
4082 assert_eq!(delivered[0].event.task_id, "long-job");
4083
4084 // Delivered and recently finished: still inside the retention window.
4085 manager.cleanup(FINISHED_SHELL_MAX_AGE);
4086 assert!(manager.inspect_job("long-job").is_ok());
4087
4088 // Delivered and finished longer ago than the window: evicted.
4089 let shell = manager.processes.get_mut("long-job").expect("record");
4090 shell.finished_at = Instant::now().checked_sub(long_run);
4091 manager.cleanup(FINISHED_SHELL_MAX_AGE);
4092 assert!(manager.inspect_job("long-job").is_err());
4093 }
4094
4095 /// Nothing ever drains a completion owned by a Runtime API scope or by a
4096 /// session that is no longer active, so an undelivered job must still age out
4097 /// (counted from when it finished) instead of staying until the count cap.
4098 #[test]
4099 fn cleanup_ages_out_undelivered_completions_after_finish() {
4100 let tmp = tempdir().expect("tempdir");
4101 let mut manager = ShellManager::new(tmp.path().to_path_buf());
4102 let max_age = Duration::from_secs(30);
4103 manager.seed_finished_record_for_test("api-job", Duration::from_secs(50));
4104 let shell = manager.processes.get_mut("api-job").expect("record");
4105 shell.owner_session_id = "api:thread-1".to_string();
4106 shell.finished_at = Instant::now().checked_sub(Duration::from_secs(40));
4107 assert!(!shell.completion_reported);
4108
4109 manager.cleanup(max_age);
4110 assert!(
4111 manager.inspect_job("api-job").is_err(),
4112 "an undelivered completion finished longer ago than max_age must age out"
4113 );
4114 }
4115
4116 /// The count ceiling evicts the records that finished longest ago, the same
4117 /// clock the age rule uses: a long job that just finished is not "oldest".
4118 #[test]
4119 fn finished_job_bounds_evict_by_finish_time() {
4120 let tmp = tempdir().expect("tempdir");
4121 let mut manager = ShellManager::new(tmp.path().to_path_buf());
4122 let now = Instant::now();
4123 for index in 0..MAX_FINISHED_SHELL_RECORDS {
4124 let id = format!("short-{index:03}");
4125 // Started 20 s ago, finished 10 s ago.
4126 manager.seed_finished_record_for_test(id.clone(), Duration::from_secs(20));
4127 let shell = manager.processes.get_mut(&id).expect("record");
4128 shell.finished_at = now.checked_sub(Duration::from_secs(10));
4129 shell.completion_reported = true;
4130 }
4131 // Started 60 s ago, finished just now.
4132 manager.seed_finished_record_for_test("long-job", Duration::from_secs(60));
4133 manager
4134 .processes
4135 .get_mut("long-job")
4136 .expect("record")
4137 .completion_reported = true;
4138
4139 manager.cleanup(FINISHED_SHELL_MAX_AGE);
4140 assert_eq!(manager.tracked_job_count(), MAX_FINISHED_SHELL_RECORDS);
4141 assert!(
4142 manager.inspect_job("long-job").is_ok(),
4143 "the most recently finished job was evicted first"
4144 );
4145 }
4146
4147 /// The synchronous path closes stdin when there is no input: `cat` gets EOF
4148 /// instead of reading, or blocking on, Codewhale's own stdin.
4149 #[cfg(unix)]
4150 #[test]
4151 fn sync_command_without_input_gets_eof_not_inherited_stdin() {
4152 let tmp = tempdir().expect("tempdir");
4153 let mut manager = ShellManager::new(tmp.path().to_path_buf());
4154 let started = Instant::now();
4155 let result = manager
4156 .execute_with_options_env(
4157 "cat; echo after-eof",
4158 None,
4159 15_000,
4160 false,
4161 None,
4162 false,
4163 None,
4164 HashMap::new(),
4165 )
4166 .expect("run");
4167 assert!(
4168 started.elapsed() < Duration::from_secs(10),
4169 "stdin read blocked for {:?}",
4170 started.elapsed()
4171 );
4172 assert!(result.stdout.contains("after-eof"), "{}", result.stdout);
4173 }
4174
4175 /// A backgrounded lowercase `bash` job: each read returns only output the
4176 /// caller has not seen, not the whole retained tail again.
4177 #[cfg(unix)]
4178 #[test]
4179 fn bounded_job_delta_returns_only_new_output() {
4180 let tmp = tempdir().expect("tempdir");
4181 let mut manager = ShellManager::new(tmp.path().to_path_buf());
4182 let started = manager
4183 .execute_with_options_env_for_owner_and_work(
4184 "printf 'first-chunk\\n'; while [ ! -e go ]; do sleep 0.05; done; printf 'second-chunk\\n'",
4185 None,
4186 60_000,
4187 true,
4188 None,
4189 false,
4190 None,
4191 HashMap::new(),
4192 None,
4193 String::new(),
4194 None,
4195 None,
4196 None,
4197 None,
4198 false,
4199 (1, BASH_MAX_TIMEOUT_MS),
4200 )
4201 .expect("spawn bounded job");
4202 let task_id = started.task_id.expect("task id");
4203 assert!(
4204 manager.processes[&task_id].bounded_output.is_some(),
4205 "fixture must exercise the bounded accumulator"
4206 );
4207
4208 let deadline = Instant::now() + Duration::from_secs(10);
4209 let mut seen = String::new();
4210 while !seen.contains("first-chunk") {
4211 assert!(Instant::now() < deadline, "first chunk never arrived");
4212 let delta = manager
4213 .get_output_delta(&task_id, false, 0)
4214 .expect("first delta");
4215 seen.push_str(&delta.result.stdout);
4216 std::thread::sleep(Duration::from_millis(20));
4217 }
4218 assert_eq!(seen.matches("first-chunk").count(), 1, "{seen}");
4219 // The fixture writes its second chunk only now, so the first delta cannot
4220 // have raced past it.
4221 std::fs::write(tmp.path().join("go"), "").expect("release fixture");
4222
4223 let rest = manager
4224 .get_output_delta(&task_id, true, 10_000)
4225 .expect("remaining delta");
4226 assert_ne!(rest.result.status, ShellStatus::Running);
4227 assert!(
4228 rest.result.stdout.contains("second-chunk"),
4229 "{}",
4230 rest.result.stdout
4231 );
4232 assert!(
4233 !rest.result.stdout.contains("first-chunk"),
4234 "a delta must not repeat output already returned: {:?}",
4235 rest.result.stdout
4236 );
4237 }
4238
4239 /// A foreground command that reads stdin gets EOF instead of blocking until
4240 /// the timeout kills it.
4241 #[cfg(unix)]
4242 #[tokio::test]
4243 async fn foreground_command_reading_stdin_gets_eof() {
4244 let tmp = tempdir().expect("tempdir");
4245 let ctx = ToolContext::new(tmp.path());
4246 let started = Instant::now();
4247 let result = BashTool::new("Bash")
4248 .execute(
4249 json!({"command": "cat; echo after-eof", "timeout_ms": 15_000}),
4250 &ctx,
4251 )
4252 .await
4253 .expect("execute");
4254 assert!(
4255 started.elapsed() < Duration::from_secs(10),
4256 "stdin read blocked for {:?}",
4257 started.elapsed()
4258 );
4259 assert!(result.content.contains("after-eof"), "{}", result.content);
4260 }
4261
4262 /// Regression for #1691: a `git commit -m "feat: complete sub-pages"` shell
4263 /// command must reach the OS shell with its quoted message intact (one argv
4264 /// slot), never split into `feat:` / `complete` / `sub-pages"`.
4265 #[test]
4266 fn issue_1691_quoted_commit_message_round_trips() {
4267 let cmd = r#"git commit -m "feat: complete sub-pages""#;
4268 let spec = CommandSpec::shell(
4269 cmd,
4270 std::path::PathBuf::from("/tmp"),
4271 Duration::from_secs(5),
4272 );
4273
4274 let dispatcher = crate::shell_dispatcher::global_dispatcher();
4275 // The whole command (with quotes) is a single argv entry. The actual
4276 // shell binary can vary by platform — and the dispatcher may wrap the
4277 // payload (encoding prefix, exit-code capture) — but the payload itself
4278 // must stay intact in ONE shell arg. We never split the command string
4279 // ourselves. This single-line ASCII command never takes the PowerShell
4280 // temp `-File` path, so the payload stays on the argv.
4281 assert_eq!(spec.program, dispatcher.kind().binary());
4282 let carriers = spec
4283 .args
4284 .iter()
4285 .filter(|arg| arg.contains(r#""feat: complete sub-pages""#))
4286 .count();
4287 assert_eq!(carriers, 1, "args: {:?}", spec.args);
4288 assert!(
4289 !spec
4290 .args
4291 .iter()
4292 .any(|arg| arg == "feat:" || arg == "complete" || arg == "sub-pages\""),
4293 "args: {:?}",
4294 spec.args
4295 );
4296 assert_eq!(spec.display_command(), cmd);
4297
4298 let mut built = Command::new(&spec.program);
4299 push_shell_args(&mut built, &spec.program, &spec.args);
4300 let got: Vec<String> = built
4301 .get_args()
4302 .map(|a| a.to_string_lossy().into_owned())
4303 .collect();
4304 assert_eq!(got, spec.args);
4305 }
4306
4307 /// When no `cwd` is provided, the shell should run in `context.workspace`,
4308 /// not in the ShellManager's default_workspace. This ensures sub-agents in
4309 /// worktrees run commands in the worktree directory rather than the parent.
4310 ///
4311 /// Without the `context.workspace` default (stashed): runs in sm_dir → FAILS
4312 /// With the `context.workspace` default (unstashed): runs in ctx_dir → PASSES
4313 #[tokio::test]
4314 async fn default_cwd_uses_context_workspace_not_shell_manager_default() {
4315 let ctx_dir = tempdir().expect("ctx tempdir");
4316 let sm_dir = tempdir().expect("sm tempdir");
4317
4318 // Create distinct dirs — write a marker in each so we can tell them apart.
4319 std::fs::write(ctx_dir.path().join("I_AM_CTX_DIR"), "").unwrap();
4320 std::fs::write(sm_dir.path().join("I_AM_SM_DIR"), "").unwrap();
4321
4322 // ToolContext whose workspace is ctx_dir...
4323 let ctx = ToolContext::new(ctx_dir.path())
4324 // ...but whose ShellManager's default_workspace is sm_dir.
4325 .with_shell_manager(new_shared_shell_manager(sm_dir.path().to_path_buf()));
4326
4327 // Assert directory identity through marker files instead of comparing the
4328 // shell's printed path. PowerShell and `canonicalize` can spell the same
4329 // Windows path differently (for example, with a verbatim-path prefix).
4330 let command = if cfg!(windows) {
4331 "if (Test-Path -LiteralPath 'I_AM_CTX_DIR') { Write-Output 'context-workspace' } elseif (Test-Path -LiteralPath 'I_AM_SM_DIR') { Write-Output 'manager-workspace' } else { Write-Output 'missing-workspace' }"
4332 } else {
4333 "if [ -f I_AM_CTX_DIR ]; then printf 'context-workspace'; elif [ -f I_AM_SM_DIR ]; then printf 'manager-workspace'; else printf 'missing-workspace'; fi"
4334 };
4335 let result = BashTool::new("Bash")
4336 .execute(json!({"command": command}), &ctx)
4337 .await
4338 .expect("shell execute");
4339 assert!(result.success, "command failed: {:?}", result.content);
4340
4341 assert!(
4342 result
4343 .content
4344 .lines()
4345 .any(|line| line.trim() == "context-workspace"),
4346 "expected context.workspace marker, but shell reported: {:?}",
4347 result.content
4348 );
4349 }
4350
4351 // ── Kill-path overshoot regression tests (FINISH-0.9.4 #52 multiplier 2) ─────
4352 //
4353 // The foreground Bash kill path must return at ~timeout + a small bounded
4354 // grace, even when the command ignores SIGTERM or a descendant escapes the
4355 // process group while holding the output pipe open. Before the fix, an
4356 // escaped descendant wedged the blocking reader-thread join inside kill()
4357 // until the descendant exited on its own (observed: ~180s past a 120s
4358 // timeout in the wild).
4359
4360 #[cfg(unix)]
4361 const SHELL_SIGTERM_HELPER_ENV: &str = "CODEWHALE_SHELL_SIGTERM_HELPER";
4362 #[cfg(unix)]
4363 const SHELL_ESCAPE_HELPER_ENV: &str = "CODEWHALE_SHELL_ESCAPE_HELPER";
4364 #[cfg(unix)]
4365 const SHELL_ESCAPED_GRANDCHILD_ENV: &str = "CODEWHALE_SHELL_ESCAPED_GRANDCHILD";
4366
4367 /// Helper role: ignore SIGTERM and idle. Runs as the shell's direct child
4368 /// (same process group), so only the SIGKILL escalation can stop it.
4369 #[cfg(unix)]
4370 #[test]
4371 fn shell_sigterm_ignoring_helper_process() {
4372 if std::env::var(SHELL_SIGTERM_HELPER_ENV).ok().as_deref() != Some("1") {
4373 return;
4374 }
4375 unsafe {
4376 libc::signal(libc::SIGTERM, libc::SIG_IGN);
4377 }
4378 let pid_file = PathBuf::from(
4379 std::env::var(SHELL_DESCENDANT_PID_FILE_ENV).expect("sigterm helper pid file"),
4380 );
4381 std::fs::write(pid_file, std::process::id().to_string()).expect("write sigterm helper pid");
4382 loop {
4383 std::thread::sleep(Duration::from_millis(10));
4384 }
4385 }
4386
4387 /// Helper role: spawn a grandchild in its OWN process group (escaping the
4388 /// shell's group) that inherits the output pipe, then exit immediately. The
4389 /// wrapper shell keeps running (`sleep` after `&`), so the job stays Running
4390 /// while the escaped grandchild holds the reader thread's pipe open.
4391 #[cfg(unix)]
4392 // The grandchild deliberately outlives this helper and is never wait()ed on —
4393 // escaping reaping is exactly what the regression exercises; the test reaps
4394 // it directly via SIGKILL at the end.
4395 #[allow(clippy::zombie_processes)]
4396 #[test]
4397 fn shell_group_escape_helper_process() {
4398 if std::env::var(SHELL_ESCAPE_HELPER_ENV).ok().as_deref() != Some("1") {
4399 return;
4400 }
4401 let test_binary = std::env::current_exe().expect("current test binary");
4402 let pid_file = std::env::var(SHELL_DESCENDANT_PID_FILE_ENV).expect("escape pid file");
4403 let mut cmd = Command::new(test_binary);
4404 cmd.arg("--exact")
4405 .arg("tools::shell::tests::shell_escaped_grandchild_helper_process")
4406 .arg("--nocapture")
4407 .env(SHELL_ESCAPED_GRANDCHILD_ENV, "1")
4408 .env(SHELL_DESCENDANT_PID_FILE_ENV, pid_file);
4409 // A distinct process group is enough to escape `kill(-wrapper_pgid)`;
4410 // stdout/stderr are inherited, so the grandchild keeps the pipe open.
4411 #[cfg(unix)]
4412 cmd.process_group(0);
4413 let _child = cmd.spawn().expect("spawn escaped grandchild");
4414 }
4415
4416 /// Helper role: the escaped grandchild — ignores SIGTERM, reports its pid,
4417 /// then idles (holding the inherited output pipe open the whole time).
4418 #[cfg(unix)]
4419 #[test]
4420 fn shell_escaped_grandchild_helper_process() {
4421 if std::env::var(SHELL_ESCAPED_GRANDCHILD_ENV).ok().as_deref() != Some("1") {
4422 return;
4423 }
4424 unsafe {
4425 libc::signal(libc::SIGTERM, libc::SIG_IGN);
4426 }
4427 let pid_file =
4428 PathBuf::from(std::env::var(SHELL_DESCENDANT_PID_FILE_ENV).expect("grandchild pid file"));
4429 std::fs::write(pid_file, std::process::id().to_string()).expect("write grandchild pid");
4430 std::thread::sleep(Duration::from_secs(30));
4431 }
4432
4433 /// Required regression: a foreground command that ignores SIGTERM must be
4434 /// dead and the tool must have returned within timeout + a small grace
4435 /// (2s timeout, assert wall < 10s).
4436 #[cfg(unix)]
4437 #[tokio::test]
4438 async fn foreground_timeout_kills_sigterm_ignoring_command_within_grace() {
4439 let tmp = tempdir().expect("tempdir");
4440 let pid_file = tmp.path().join("sigterm-helper.pid");
4441 let test_binary = std::env::current_exe().expect("current test binary");
4442 let command = format!(
4443 "{SHELL_SIGTERM_HELPER_ENV}=1 {SHELL_DESCENDANT_PID_FILE_ENV}={} exec {} --exact {} --nocapture",
4444 shell_words::quote(&pid_file.display().to_string()),
4445 shell_words::quote(&test_binary.display().to_string()),
4446 shell_words::quote("tools::shell::tests::shell_sigterm_ignoring_helper_process"),
4447 );
4448 let ctx = ToolContext::new(tmp.path());
4449
4450 let started = Instant::now();
4451 let result = BashTool::new("Bash")
4452 .execute(json!({"command": command, "timeout_ms": 2_000}), &ctx)
4453 .await
4454 .expect("execute");
4455 let wall = started.elapsed();
4456
4457 assert!(!result.success);
4458 let meta = result.metadata.expect("metadata");
4459 assert_eq!(meta.get("status").and_then(Value::as_str), Some("TimedOut"));
4460 assert!(
4461 wall < Duration::from_secs(10),
4462 "kill path overshot the 2s timeout: wall {wall:?}"
4463 );
4464 let helper_pid = wait_for_shell_pid_file(&pid_file);
4465 assert!(
4466 wait_for_shell_pid_exit(helper_pid),
4467 "SIGTERM-ignoring helper {helper_pid} survived the timeout kill"
4468 );
4469 }
4470
4471 /// Regression for the ~180s kill-path overshoot: a descendant that escaped
4472 /// the process group keeps the output pipe open after the group is killed.
4473 /// kill() must still return within a bounded grace instead of blocking on
4474 /// the reader-thread join until the descendant exits on its own.
4475 #[cfg(unix)]
4476 #[tokio::test]
4477 async fn kill_returns_promptly_when_escaped_descendant_holds_pipe_open() {
4478 let tmp = tempdir().expect("tempdir");
4479 let pid_file = tmp.path().join("escaped-grandchild.pid");
4480 let test_binary = std::env::current_exe().expect("current test binary");
4481 let command = format!(
4482 "{SHELL_ESCAPE_HELPER_ENV}=1 {SHELL_DESCENDANT_PID_FILE_ENV}={} {} --exact {} --nocapture & sleep 60",
4483 shell_words::quote(&pid_file.display().to_string()),
4484 shell_words::quote(&test_binary.display().to_string()),
4485 shell_words::quote("tools::shell::tests::shell_group_escape_helper_process"),
4486 );
4487 let mut manager = ShellManager::new(tmp.path().to_path_buf());
4488 let started_bg =
4489 execute_shell(&mut manager, &command, None, 600_000, true).expect("start wrapper");
4490 let task_id = started_bg.task_id.expect("task id");
4491 let grandchild = wait_for_shell_pid_file(&pid_file);
4492
4493 let started = Instant::now();
4494 let killed = manager.kill(&task_id).expect("kill");
4495 let wall = started.elapsed();
4496
4497 assert_eq!(killed.status, ShellStatus::Killed);
4498 assert!(
4499 wall < Duration::from_secs(10),
4500 "kill blocked {wall:?} on a reader wedged by an escaped descendant"
4501 );
4502
4503 // Cleanup: the escaped grandchild is out of reach of the group kill by
4504 // construction; reap it directly so the test does not leak a sleeper.
4505 unsafe {
4506 libc::kill(grandchild, libc::SIGKILL);
4507 }
4508 assert!(wait_for_shell_pid_exit(grandchild));
4509 }
4510
4511 /// `Bash` was the only action wrapper whose catch-all fell through to its most
4512 /// dangerous branch: an unrecognised action ran the command instead.
4513 #[tokio::test]
4514 async fn unknown_bash_action_is_refused_instead_of_running_the_command() {
4515 let workspace = tempdir().expect("workspace");
4516 let context = ToolContext::new(workspace.path().to_path_buf());
4517 let marker = workspace.path().join("should-not-exist");
4518
4519 let error = BashTool::new("Bash")
4520 .execute(
4521 json!({
4522 "action": "kill",
4523 "command": format!("touch {}", marker.display()),
4524 }),
4525 &context,
4526 )
4527 .await
4528 .expect_err("unknown action must be refused");
4529
4530 let message = error.to_string();
4531 assert!(message.contains("Unknown Bash action"), "{message}");
4532 assert!(message.contains("kill"), "{message}");
4533 assert!(
4534 message.contains("run, wait, interact, cancel"),
4535 "must name the actions that dispatch: {message}"
4536 );
4537 assert!(!marker.exists(), "the command must not have run");
4538 }
4539
4540 /// A NUL byte cannot cross the `exec` boundary: `Command` panics on it.
4541 /// Refuse with the byte offset before anything spawns (#5529).
4542 #[tokio::test]
4543 async fn nul_byte_in_shell_command_is_refused_before_spawn() {
4544 let workspace = tempdir().expect("workspace");
4545 let context = ToolContext::new(workspace.path().to_path_buf());
4546 let marker = workspace.path().join("should-not-exist");
4547
4548 let error = BashTool::new("Bash")
4549 .execute(
4550 json!({
4551 "command": format!("echo hi\0; touch {}", marker.display()),
4552 }),
4553 &context,
4554 )
4555 .await
4556 .expect_err("NUL byte must be refused");
4557
4558 let message = error.to_string();
4559 assert!(message.contains("NUL byte"), "{message}");
4560 assert!(message.contains("byte offset 7"), "{message}");
4561 assert!(!marker.exists(), "the command must not have run");
4562 }
4563
4564 /// `cwd` crosses the same boundary via `current_dir`, so the guard covers it
4565 /// too (#5529).
4566 #[tokio::test]
4567 async fn nul_byte_in_shell_cwd_is_refused_before_spawn() {
4568 let workspace = tempdir().expect("workspace");
4569 let context = ToolContext::new(workspace.path().to_path_buf());
4570
4571 let error = BashTool::new("Bash")
4572 .execute(
4573 json!({
4574 "command": "echo hi",
4575 "cwd": "sub\0dir",
4576 }),
4577 &context,
4578 )
4579 .await
4580 .expect_err("NUL byte must be refused");
4581
4582 let message = error.to_string();
4583 assert!(message.contains("NUL byte"), "{message}");
4584 assert!(message.contains("cwd"), "{message}");
4585 assert!(message.contains("byte offset 3"), "{message}");
4586 }
4587
4588 /// The same hole one type down. `and_then(as_str).unwrap_or("run")` read a
4589 /// non-string `action` as absent and fell through to the branch that executes
4590 /// arbitrary code, so `Bash{action: 3, command: "…"}` ran the command. `File`,
4591 /// `Git`, `Web`, and `Run` all refuse a non-string action; the tool that runs
4592 /// shell commands must not be the lenient one.
4593 #[tokio::test]
4594 async fn non_string_bash_action_is_refused_instead_of_running_the_command() {
4595 let workspace = tempdir().expect("workspace");
4596 let context = ToolContext::new(workspace.path().to_path_buf());
4597
4598 for action in [json!(3), json!(true), json!(["run"]), json!({"run": true})] {
4599 let marker = workspace.path().join(format!("marker-{action}"));
4600 let error = BashTool::new("Bash")
4601 .execute(
4602 json!({
4603 "action": action,
4604 "command": format!("touch {}", marker.display()),
4605 }),
4606 &context,
4607 )
4608 .await
4609 .expect_err("a non-string action must be refused");
4610
4611 let message = error.to_string();
4612 assert!(
4613 message.contains("'action'"),
4614 "must name the parameter: {message}"
4615 );
4616 assert!(
4617 message.contains("must be a string"),
4618 "must name the expected type: {message}"
4619 );
4620 assert!(!marker.exists(), "the command must not have run: {action}");
4621 }
4622 }
4623
4624 /// The same hole for the data fields (2026-08-04 review). A non-string
4625 /// `stdin` was silently dropped — the command ran with NO stdin and reported
4626 /// success, the silent-drop failure this lane exists to close. A non-string
4627 /// `cwd` silently ran in the workspace default. And a numeric `task_id` was
4628 /// reported as "missing", steering the model's retry the wrong way.
4629 #[tokio::test]
4630 async fn wrongly_typed_stdin_cwd_and_task_id_are_refused_not_dropped() {
4631 let workspace = tempdir().expect("workspace");
4632 let context = ToolContext::new(workspace.path().to_path_buf());
4633
4634 let marker = workspace.path().join("stdin-marker");
4635 let error = BashTool::new("Bash")
4636 .execute(
4637 json!({
4638 "command": format!("touch {}", marker.display()),
4639 "stdin": 12345,
4640 }),
4641 &context,
4642 )
4643 .await
4644 .expect_err("non-string stdin must be refused, never silently dropped");
4645 let message = error.to_string();
4646 assert!(message.contains("'stdin'"), "names the field: {message}");
4647 assert!(
4648 message.contains("must be a string"),
4649 "names the expected type: {message}"
4650 );
4651 assert!(!marker.exists(), "the command must not have run");
4652
4653 let error = BashTool::new("Bash")
4654 .execute(json!({ "command": "pwd", "cwd": 123 }), &context)
4655 .await
4656 .expect_err("non-string cwd must be refused, never defaulted");
4657 assert!(error.to_string().contains("'cwd'"), "{error}");
4658
4659 let error = BashTool::new("Bash")
4660 .execute(json!({ "action": "wait", "task_id": 42 }), &context)
4661 .await
4662 .expect_err("non-string task_id is a type error");
4663 let message = error.to_string();
4664 assert!(
4665 message.contains("'task_id'") && message.contains("must be a string"),
4666 "a supplied-but-mistyped task_id must not read as missing: {message}"
4667 );
4668 }
4669
4670 /// `null` is the wire spelling of absence, and `action` documents a `run`
4671 /// default — so the strictness above must not swallow the default.
4672 #[tokio::test]
4673 async fn absent_or_null_bash_action_still_defaults_to_run() {
4674 let workspace = tempdir().expect("workspace");
4675 let context = ToolContext::new(workspace.path().to_path_buf());
4676
4677 for input in [
4678 json!({"command": "echo defaulted"}),
4679 json!({"action": null, "command": "echo defaulted"}),
4680 ] {
4681 let result = BashTool::new("Bash")
4682 .execute(input.clone(), &context)
4683 .await
4684 .unwrap_or_else(|err| panic!("{input} must still run: {err}"));
4685 assert!(result.success, "{input}: {}", result.content);
4686 assert!(result.content.contains("defaulted"), "{}", result.content);
4687 }
4688 }
4689
4690 /// Negative case for the strictness above: every legitimate action still
4691 /// dispatches to its own handler rather than the action refusal. `wait`,
4692 /// `interact`, and `cancel` are checked by the error they raise *after*
4693 /// dispatch (a missing/unknown task), which only their own handlers produce.
4694 #[tokio::test]
4695 async fn every_valid_bash_action_still_dispatches() {
4696 let workspace = tempdir().expect("workspace");
4697 let context = ToolContext::new(workspace.path().to_path_buf());
4698 let tool = BashTool::new("Bash");
4699
4700 let ran = tool
4701 .execute(
4702 json!({"action": "run", "command": "echo dispatched"}),
4703 &context,
4704 )
4705 .await
4706 .expect("action=run must dispatch");
4707 assert!(ran.success, "{}", ran.content);
4708
4709 for input in [
4710 json!({"action": "wait", "task_id": "no-such-task"}),
4711 json!({"action": "interact", "task_id": "no-such-task", "stdin": "y\n"}),
4712 json!({"action": "cancel", "task_id": "no-such-task"}),
4713 ] {
4714 let outcome = tool.execute(input.clone(), &context).await;
4715 let message = match outcome {
4716 Ok(result) => result.content,
4717 Err(err) => err.to_string(),
4718 };
4719 assert!(
4720 !message.contains("Unknown Bash action") && !message.contains("must be a string"),
4721 "{input} must reach its own handler, got: {message}"
4722 );
4723 }
4724
4725 // `cancel` with `all` needs no task at all and must stay a success.
4726 let cancelled = tool
4727 .execute(json!({"action": "cancel", "all": true}), &context)
4728 .await
4729 .expect("action=cancel all=true must dispatch");
4730 assert!(cancelled.success, "{}", cancelled.content);
4731 }
4732
4733 /// The stdin aliases were real but undocumented: a model that wrote `input`
4734 /// or `data` got them honoured with nothing in the schema saying so, and a
4735 /// maintainer reading the schema would have removed them as dead. Advertise
4736 /// them, and hold every spelling to the same behavior.
4737 #[tokio::test]
4738 async fn every_advertised_stdin_spelling_reaches_the_command() {
4739 let workspace = tempdir().expect("workspace");
4740 let context = ToolContext::new(workspace.path().to_path_buf());
4741 let schema = BashTool::new("Bash").input_schema();
4742
4743 // `cat` is Unix-only; the dispatcher runs PowerShell or `cmd` on Windows,
4744 // where it is either absent or an alias for `Get-Content`, which reads a
4745 // file and not stdin. Ask for this platform's echo-stdin spelling — the
4746 // same helper `test_write_stdin_streams_output` uses.
4747 let echo_stdin = echo_stdin_command();
4748 for spelling in ["stdin", "input", "data"] {
4749 assert!(
4750 schema["properties"][spelling].is_object(),
4751 "`{spelling}` is honoured at runtime and must be advertised"
4752 );
4753 let result = BashTool::new("Bash")
4754 .execute(
4755 json!({"command": echo_stdin, spelling: "PIPED_THROUGH_ALIAS\n"}),
4756 &context,
4757 )
4758 .await
4759 .unwrap_or_else(|err| panic!("`{spelling}` must deliver stdin: {err}"));
4760 assert!(
4761 result.content.contains("PIPED_THROUGH_ALIAS"),
4762 "`{spelling}` did not reach the command: {}",
4763 result.content
4764 );
4765 }
4766
4767 // `id` is the same undocumented shape one parameter over.
4768 assert!(
4769 schema["properties"]["id"].is_object(),
4770 "`id` is accepted for `task_id` at runtime and must be advertised"
4771 );
4772 assert!(
4773 schema["properties"]["task_id"]["description"]
4774 .as_str()
4775 .is_some_and(|text| text.contains("`id`")),
4776 "task_id must name its alias"
4777 );
4778 }
4779
4780 /// The schema declared no `required` key at all, so `Bash{}` — no command, no
4781 /// task — was schema-valid for the tool that runs shell commands. What is
4782 /// required is per-action, so it is spelled as root `anyOf` required groups,
4783 /// the same shape `finance` and `apply_patch` already use.
4784 #[test]
4785 fn bash_schema_declares_what_each_action_requires() {
4786 let schema = BashTool::new("Bash").input_schema();
4787 let groups: Vec<Vec<String>> = schema["anyOf"]
4788 .as_array()
4789 .expect("root anyOf required groups")
4790 .iter()
4791 .map(|group| {
4792 group["required"]
4793 .as_array()
4794 .expect("required group")
4795 .iter()
4796 .map(|name| name.as_str().expect("required name").to_string())
4797 .collect()
4798 })
4799 .collect();
4800
4801 for expected in [["command"], ["task_id"], ["id"], ["all"]] {
4802 assert!(
4803 groups.iter().any(|group| group.as_slice() == expected),
4804 "missing required group {expected:?} in {groups:?}"
4805 );
4806 }
4807 // A required name the same schema does not advertise would be
4808 // unsatisfiable: the model could not learn what to send.
4809 for group in &groups {
4810 for name in group {
4811 assert!(
4812 schema["properties"][name].is_object(),
4813 "`{name}` is required but not advertised"
4814 );
4815 }
4816 }
4817 }
4818
4819 /// A root `anyOf` is not portable to every provider, so prove the fallback
4820 /// the sanitizer promises: Responses/xAI drop root composition, and the
4821 /// constraint has to survive as a description note rather than vanishing.
4822 #[test]
4823 fn bash_required_groups_survive_a_provider_that_drops_root_composition() {
4824 let mut schema = BashTool::new("Bash").input_schema();
4825 let note = crate::tools::schema_sanitize::sanitize_for_responses(&mut schema)
4826 .expect("dropped required groups must be restated for the model");
4827
4828 assert!(note.contains("At least one"), "{note}");
4829 for name in ["`command`", "`task_id`", "`id`", "`all`"] {
4830 assert!(note.contains(name), "note must name {name}: {note}");
4831 }
4832 assert_eq!(schema["type"], "object");
4833 assert!(schema.get("anyOf").is_none(), "root anyOf must be removed");
4834 assert!(schema["properties"]["command"].is_object());
4835 }
4836
4837 /// Every hint in this file has to name a tool the model can actually call.
4838 /// `exec_shell` / `exec_shell_wait` were retired in v0.9.3.
4839 #[test]
4840 fn shell_recovery_hints_name_only_dispatchable_tools() {
4841 assert!(!FOREGROUND_TIMEOUT_RECOVERY_HINT.contains("exec_shell"));
4842 assert!(FOREGROUND_TIMEOUT_RECOVERY_HINT.contains("Bash"));
4843 assert!(FOREGROUND_TIMEOUT_RECOVERY_HINT.contains("action=\"wait\""));
4844 }
4845
4846 /// One documented default hid three real ones: `wait` uses 30s and
4847 /// `interact` 1s, so a model omitting `timeout_ms` on `wait` got a quarter of
4848 /// the timeout the schema promised.
4849 #[test]
4850 fn timeout_ms_description_covers_every_action_default() {
4851 let schema = BashTool::new("Bash").input_schema();
4852 let description = schema["properties"]["timeout_ms"]["description"]
4853 .as_str()
4854 .expect("timeout_ms description");
4855
4856 for expected in ["120000", "600000", "30000", "1000"] {
4857 assert!(
4858 description.contains(expected),
4859 "missing {expected}: {description}"
4860 );
4861 }
4862 }
4863
4864 #[cfg(unix)]
4865 fn authorized_persistent_service_context(workspace: &Path) -> ToolContext {
4866 let mut context = ToolContext::new(workspace.to_path_buf())
4867 .with_elevated_sandbox_policy(ExecutionSandboxPolicy::DangerFullAccess);
4868 context.persist_services_enabled = true;
4869 context.tool_authority = None;
4870 context.shell_policy = ShellPolicy::Full;
4871 context.auto_approve = true;
4872 context
4873 }
4874
4875 #[cfg(unix)]
4876 fn persistent_service_test_lock() -> &'static tokio::sync::Mutex<()> {
4877 static LOCK: OnceLock<tokio::sync::Mutex<()>> = OnceLock::new();
4878 LOCK.get_or_init(|| tokio::sync::Mutex::new(()))
4879 }
4880
4881 #[cfg(unix)]
4882 #[tokio::test]
4883 async fn persistent_service_requires_explicit_headless_exec_authority() {
4884 let workspace = tempdir().expect("workspace");
4885 let context = ToolContext::new(workspace.path().to_path_buf());
4886
4887 let error = BashTool::new("Bash")
4888 .execute(
4889 json!({
4890 "action": "run",
4891 "command": "sleep 30",
4892 "background": true,
4893 "persist": true,
4894 }),
4895 &context,
4896 )
4897 .await
4898 .expect_err("ordinary tool contexts must reject ownership transfer");
4899
4900 assert!(error.to_string().contains("real headless `codewhale exec`"));
4901 }
4902
4903 #[cfg(unix)]
4904 #[tokio::test]
4905 async fn committed_persistent_service_survives_manager_drop_and_reports_identity() {
4906 let _guard = persistent_service_test_lock().lock().await;
4907 let workspace = tempdir().expect("workspace");
4908 let marker = workspace.path().join("persistent-service-finished");
4909 let context = authorized_persistent_service_context(workspace.path());
4910
4911 let result = BashTool::new("Bash")
4912 .execute(
4913 json!({
4914 "action": "run",
4915 "command": format!("sleep 1; printf released > '{}'", marker.display()),
4916 "background": true,
4917 "persist": true,
4918 }),
4919 &context,
4920 )
4921 .await
4922 .expect("stage persistent service");
4923 assert!(result.success, "{result:?}");
4924 assert_eq!(
4925 result
4926 .metadata
4927 .as_ref()
4928 .and_then(|metadata| metadata["ownership"].as_str()),
4929 Some("managed_pending_exec_success")
4930 );
4931 let task_id = result.metadata.as_ref().unwrap()["task_id"]
4932 .as_str()
4933 .expect("task id")
4934 .to_string();
4935
4936 let receipts = context
4937 .shell_manager
4938 .lock()
4939 .expect("shell manager")
4940 .commit_persistent_services()
4941 .expect("commit persistent service");
4942 assert_eq!(receipts.len(), 1);
4943 assert_eq!(receipts[0].task_id, task_id);
4944 assert_eq!(receipts[0].process_group_id, receipts[0].pid);
4945 assert_eq!(receipts[0].ownership, "external");
4946
4947 drop(context);
4948 let deadline = Instant::now() + Duration::from_secs(5);
4949 while !marker.exists() && Instant::now() < deadline {
4950 std::thread::sleep(Duration::from_millis(25));
4951 }
4952 assert!(
4953 marker.exists(),
4954 "released service must survive Codewhale manager teardown"
4955 );
4956 }
4957
4958 #[cfg(unix)]
4959 #[tokio::test]
4960 async fn signal_cleanup_kills_staged_persistent_service_group() {
4961 let _guard = persistent_service_test_lock().lock().await;
4962 let workspace = tempdir().expect("workspace");
4963 let context = authorized_persistent_service_context(workspace.path());
4964
4965 let result = BashTool::new("Bash")
4966 .execute(
4967 json!({
4968 "action": "run",
4969 "command": "sleep 30",
4970 "background": true,
4971 "persist": true,
4972 }),
4973 &context,
4974 )
4975 .await
4976 .expect("stage persistent service");
4977 let task_id = result.metadata.as_ref().unwrap()["task_id"]
4978 .as_str()
4979 .expect("task id")
4980 .to_string();
4981 let pid = context
4982 .shell_manager
4983 .lock()
4984 .expect("shell manager")
4985 .processes[&task_id]
4986 .child
4987 .as_ref()
4988 .and_then(ShellChild::process_id)
4989 .expect("persistent process id");
4990
4991 abort_pending_persistent_process_groups_for_exit();
4992 let pid = i32::try_from(pid).expect("pid fits pid_t");
4993 let deadline = Instant::now() + Duration::from_secs(2);
4994 let status = loop {
4995 let mut status = 0;
4996 // SAFETY: `pid` is the direct child owned by this test's manager; the
4997 // nonblocking wait only reaps that exact process.
4998 let waited = unsafe { libc::waitpid(pid, &mut status, libc::WNOHANG) };
4999 if waited == pid {
5000 break status;
5001 }
5002 assert!(
5003 Instant::now() < deadline,
5004 "signal cleanup must kill the staged service process group"
5005 );
5006 std::thread::sleep(Duration::from_millis(25));
5007 };
5008 assert!(libc::WIFSIGNALED(status));
5009 assert_eq!(libc::WTERMSIG(status), libc::SIGKILL);
5010 }
5011
5012 // === #5472: in-memory retention must be bounded ===
5013
5014 /// ~1.1 MB on stdout, fast: 30,000 lines of 37 bytes.
5015 #[cfg(unix)]
5016 fn chatty_command() -> String {
5017 "yes 0123456789abcdefghijklmnopqrstuvwxyz | head -n 30000".to_string()
5018 }
5019
5020 /// The finding-1 regression: before this bound, a single uppercase `Bash` call
5021 /// left its entire stdout resident in `ShellManager.processes` until the 1 h
5022 /// `cleanup`, and only if the user happened to open the jobs panel.
5023 #[cfg(unix)]
5024 #[tokio::test]
5025 async fn foreground_bash_releases_its_output_once_the_result_is_returned() {
5026 let tmp = tempdir().expect("tempdir");
5027 let ctx = ToolContext::new(tmp.path());
5028 let result = BashTool::new("Bash")
5029 .execute(json!({"command": chatty_command()}), &ctx)
5030 .await
5031 .expect("run chatty foreground command");
5032
5033 // The result itself is unaffected: still the same 30 KB truncation.
5034 assert!(
5035 result
5036 .content
5037 .contains("0123456789abcdefghijklmnopqrstuvwxyz")
5038 );
5039
5040 let manager = ctx.shell_manager.lock().expect("shell manager");
5041 let retained = manager.retained_output_bytes_total();
5042 assert!(
5043 retained <= RAW_STREAM_SETTLED_TAIL_BYTES * 2,
5044 "a finished foreground call must not keep its full stdout resident: \
5045 {retained} bytes still held (bound {})",
5046 RAW_STREAM_SETTLED_TAIL_BYTES * 2
5047 );
5048 }
5049
5050 /// Releasing memory must not rewrite history: the job panel keeps reporting how
5051 /// much the command actually printed.
5052 #[cfg(unix)]
5053 #[tokio::test]
5054 async fn released_output_still_reports_the_real_stream_length() {
5055 let tmp = tempdir().expect("tempdir");
5056 let ctx = ToolContext::new(tmp.path());
5057 BashTool::new("Bash")
5058 .execute(json!({"command": chatty_command()}), &ctx)
5059 .await
5060 .expect("run chatty foreground command");
5061
5062 let mut manager = ctx.shell_manager.lock().expect("shell manager");
5063 let jobs = manager.list_jobs();
5064 let job = jobs.first().expect("the finished job is still listed");
5065 assert!(
5066 job.stdout_len >= 1_000_000,
5067 "stdout_len must stay honest after release, got {}",
5068 job.stdout_len
5069 );
5070 assert!(
5071 !job.stdout_tail.is_empty(),
5072 "a diagnostic tail must survive the release"
5073 );
5074 }
5075
5076 /// A background job's bytes become a durable session artifact at drain time;
5077 /// keeping a second copy in the manager afterwards is the pure-waste term.
5078 #[cfg(unix)]
5079 #[tokio::test]
5080 async fn draining_completion_evidence_releases_the_retained_copy() {
5081 let tmp = tempdir().expect("tempdir");
5082 let ctx = ToolContext::new(tmp.path());
5083 let started = BashTool::new("Bash")
5084 .execute(
5085 json!({"command": chatty_command(), "background": true}),
5086 &ctx,
5087 )
5088 .await
5089 .expect("start background");
5090 let task_id = started
5091 .metadata
5092 .as_ref()
5093 .and_then(|metadata| metadata.get("task_id"))
5094 .and_then(Value::as_str)
5095 .expect("task id")
5096 .to_string();
5097
5098 let mut manager = ctx.shell_manager.lock().expect("shell manager");
5099 let completed = wait_for_completed_shell(&mut manager, &task_id);
5100 assert_ne!(completed.status, ShellStatus::Running);
5101
5102 let evidence = manager.drain_finished_jobs_with_evidence();
5103 assert_eq!(evidence.len(), 1);
5104 assert!(
5105 evidence[0].event.stdout_len >= 1_000_000,
5106 "the event still reports the full length"
5107 );
5108 let payload: serde_json::Value =
5109 serde_json::from_slice(&evidence[0].artifact_bytes()).expect("evidence JSON");
5110 assert!(
5111 payload["stdout"]["content"]
5112 .as_str()
5113 .expect("stdout content")
5114 .len()
5115 >= 1_000_000,
5116 "the artifact carries the exact bytes; only the manager's copy is dropped"
5117 );
5118
5119 let retained = manager.retained_output_bytes_total();
5120 assert!(
5121 retained <= RAW_STREAM_SETTLED_TAIL_BYTES * 2,
5122 "{retained} bytes still held after the evidence was published"
5123 );
5124 }
5125
5126 /// Age was the only bound, and it only ran from `list_jobs()`. Hundreds of
5127 /// finished records inside one hour were all retained.
5128 #[test]
5129 fn cleanup_bounds_finished_records_by_count() {
5130 let tmp = tempdir().expect("tempdir");
5131 let mut manager = ShellManager::new(tmp.path().to_path_buf());
5132 let seeded_count = MAX_FINISHED_SHELL_RECORDS + 40;
5133 for index in 0..seeded_count {
5134 // Give every fixture a deterministic ordering while keeping all of
5135 // them far younger than the age ceiling. Lower ids are older.
5136 manager.seed_finished_record_for_test(
5137 format!("record-{index}"),
5138 Duration::from_millis((seeded_count - index) as u64),
5139 );
5140 }
5141 manager.cleanup(FINISHED_SHELL_MAX_AGE);
5142 assert_eq!(manager.tracked_job_count(), MAX_FINISHED_SHELL_RECORDS);
5143 assert!(
5144 manager.inspect_job("record-39").is_err(),
5145 "the oldest overflow record must be evicted"
5146 );
5147 assert!(
5148 manager.inspect_job("record-40").is_ok(),
5149 "the first record inside the cap must survive"
5150 );
5151 assert!(
5152 manager
5153 .inspect_job(&format!("record-{}", seeded_count - 1))
5154 .is_ok(),
5155 "the newest record must survive"
5156 );
5157 }
5158
5159 /// #5478 nit: `/jobs` reported `2m 07s` for a 12-second command, because
5160 /// elapsed was `started_at.elapsed()` even after the job finished. A completed
5161 /// job reports the duration it finished with.
5162 #[test]
5163 fn a_finished_job_reports_its_duration_not_a_growing_elapsed() {
5164 let tmp = tempdir().expect("tempdir");
5165 let mut manager = ShellManager::new(tmp.path().to_path_buf());
5166 let result = manager
5167 .execute_with_options_env(
5168 &echo_command("frozen-elapsed"),
5169 None,
5170 10_000,
5171 true,
5172 None,
5173 false,
5174 None,
5175 std::collections::HashMap::new(),
5176 )
5177 .expect("spawn");
5178 let task_id = result.task_id.expect("task id");
5179 let completed = wait_for_completed_shell(&mut manager, &task_id);
5180 assert_ne!(completed.status, ShellStatus::Running);
5181
5182 let first = manager
5183 .list_jobs()
5184 .into_iter()
5185 .find(|job| job.id == task_id)
5186 .expect("job listed")
5187 .elapsed_ms;
5188
5189 std::thread::sleep(Duration::from_millis(400));
5190
5191 let second = manager
5192 .list_jobs()
5193 .into_iter()
5194 .find(|job| job.id == task_id)
5195 .expect("job still listed")
5196 .elapsed_ms;
5197
5198 assert_eq!(
5199 first, second,
5200 "a finished job's elapsed must stop moving; it read {first}ms then {second}ms"
5201 );
5202 assert!(
5203 second < 10_000,
5204 "the frozen value must be the real duration, not the timeout: {second}ms"
5205 );
5206 }
5207
5208 #[cfg(unix)]
5209 #[tokio::test]
5210 async fn readonly_pipeline_preserves_arguments_and_disables_git_helpers() {
5211 const PROBE: &str = "CODEWHALE_TEST_READONLY_PIPELINE_SHELL";
5212 if std::env::var_os(PROBE).is_none() {
5213 // The dispatcher is process-pinned. Exercise both the supported shell
5214 // and the fail-closed POSIX fallback without inheriting the CI shell.
5215 for shell in ["/bin/bash", "/bin/sh"] {
5216 let output = std::process::Command::new(std::env::current_exe().unwrap())
5217 .args([
5218 "--exact",
5219 "tools::shell::tests::readonly_pipeline_preserves_arguments_and_disables_git_helpers",
5220 "--test-threads=1",
5221 ])
5222 .env(PROBE, shell)
5223 .env("SHELL", shell)
5224 .output()
5225 .unwrap();
5226 assert!(
5227 output.status.success(),
5228 "read-only pipeline probe failed ({shell})\n{}\n{}",
5229 String::from_utf8_lossy(&output.stdout),
5230 String::from_utf8_lossy(&output.stderr)
5231 );
5232 }
5233 return;
5234 }
5235 let workspace = tempdir().unwrap();
5236 let outside = tempdir().unwrap();
5237 let sentinel = outside.path().join("secret");
5238 std::fs::write(&sentinel, "private-marker\n").unwrap();
5239 std::os::unix::fs::symlink(&sentinel, workspace.path().join("linked-secret")).unwrap();
5240 std::fs::write(workspace.path().join("input.txt"), "hello\n").unwrap();
5241 for name in ["-i", "-e", "-f", "--output=changed", "-o"] {
5242 std::fs::write(workspace.path().join(name), "option-shaped filename\n").unwrap();
5243 }
5244 let ctx = ToolContext::new(workspace.path())
5245 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
5246 let tool = BashTool::new("Bash");
5247 for command in ["sort * | cat", "sed -n 1p * | cat", "cat * | cat"] {
5248 // #6675: a word-leading unquoted `*` is refused before anything runs
5249 // (its matches could be option-shaped names like the ones above).
5250 let result = tool.execute(json!({"command": command}), &ctx).await;
5251 match result {
5252 Err(error) => assert!(
5253 error.to_string().contains("unquoted `*`"),
5254 "{command}: {error}"
5255 ),
5256 Ok(result) => panic!("{command} must be refused, ran: {}", result.content),
5257 }
5258 assert_eq!(
5259 std::fs::read_to_string(&sentinel).unwrap(),
5260 "private-marker\n"
5261 );
5262 assert!(!workspace.path().join("changed").exists());
5263 }
5264 let ordinary = tool
5265 .execute(json!({"command": "cat input.txt | wc -l"}), &ctx)
5266 .await
5267 .unwrap();
5268 if std::env::var(PROBE).as_deref() == Ok("/bin/sh") {
5269 assert!(!ordinary.success);
5270 assert!(
5271 ordinary
5272 .content
5273 .contains("read-only pipelines and chains require bash or zsh")
5274 );
5275 return;
5276 }
5277 assert!(ordinary.success, "{}", ordinary.content);
5278 assert!(
5279 tool.execute(json!({"command": "cat linked-secret | cat"}), &ctx)
5280 .await
5281 .is_err()
5282 );
5283 let pipeline = hardened_readonly_script("git show HEAD | cat", workspace.path()).unwrap();
5284 assert!(pipeline.contains("--no-ext-diff"));
5285 assert!(pipeline.contains("--no-textconv"));
5286 assert!(pipeline.contains("--no-show-signature"));
5287 }
5288
5289 #[tokio::test]
5290 async fn readonly_sed_extra_options_never_mutate_files() {
5291 let workspace = tempdir().unwrap();
5292 let source = workspace.path().join("input.txt");
5293 std::fs::write(&source, "first\nsecond\n").unwrap();
5294 let ctx = ToolContext::new(workspace.path())
5295 .with_shell_policy(crate::worker_profile::ShellPolicy::ReadOnly);
5296 for command in [
5297 "sed -n 1p input.txt -i",
5298 "sed -n 1p -i.bak input.txt",
5299 "sed -n 1p -e 1e input.txt",
5300 "sed -n 1p -f script input.txt",
5301 ] {
5302 let error = BashTool::new("Bash")
5303 .execute(json!({"command": command}), &ctx)
5304 .await
5305 .expect_err("refused");
5306 assert!(
5307 error.to_string().contains("[shell.readonly.command]"),
5308 "{command}: {error}"
5309 );
5310 assert_eq!(std::fs::read_to_string(&source).unwrap(), "first\nsecond\n");
5311 }
5312 let result = BashTool::new("Bash")
5313 .execute(json!({"command": "sed -n 1p input.txt"}), &ctx)
5314 .await
5315 .unwrap();
5316 assert!(result.success, "{}", result.content);
5317 }
5318
5319 /// A transiently busy Work-graph must not veto the command.
5320 ///
5321 /// `register_operation` acquires the To-do/Plan locks with a short try-lock
5322 /// spin. Before this guard existed, a shell call landing in that window failed
5323 /// outright with "To-do state is busy; operation was not registered" — observed
5324 /// twice in one live session, each time right after another tool call. The
5325 /// registration is the same bookkeeping whose `observe` half is already
5326 /// best-effort, so a busy state now degrades to an unbound run.
5327 #[tokio::test]
5328 async fn busy_work_graph_degrades_the_spawn_intent_instead_of_failing_it() {
5329 use crate::tools::plan::new_shared_plan_state;
5330 use crate::tools::todo::new_shared_todo_list;
5331 use crate::work_graph::new_shared_work_runtime;
5332
5333 let todos = new_shared_todo_list();
5334 let plan = new_shared_plan_state();
5335 let lifecycle = || ShellWorkLifecycle {
5336 work: new_shared_work_runtime(todos.clone(), plan.clone()),
5337 session_id: "session-test".to_string(),
5338 };
5339
5340 // Control: with the graph free, the intent binds.
5341 let bound = ShellSpawnIntentGuard::new(Some(lifecycle()), "shell_free", "echo hi");
5342 assert!(
5343 bound.lifecycle.is_some(),
5344 "a free work-graph must bind the spawn intent"
5345 );
5346
5347 // Busy: hold the To-do lock so the try-lock spin cannot win.
5348 let _held = todos.lock().await;
5349 // The raw register call still reports the busy state — this is exactly what
5350 // used to propagate out of the spawn path and fail the command.
5351 assert!(
5352 lifecycle()
5353 .register("shell_busy_direct", "echo hi")
5354 .is_err(),
5355 "the raw register call must observe the held lock as busy"
5356 );
5357 let busy = ShellSpawnIntentGuard::new(Some(lifecycle()), "shell_busy", "echo hi");
5358 assert!(
5359 busy.lifecycle.is_none(),
5360 "a busy work-graph must degrade to an unbound guard, not fail the spawn"
5361 );
5362 }
5363
5364 /// #6435: the guard went unbound, but its sibling handle still published the
5365 /// spawn to the unregistered operation, killed the child and reported
5366 /// "operation binding shell:<id> is not registered". A busy Work-graph must
5367 /// let the command run end to end.
5368 #[cfg(unix)]
5369 #[tokio::test]
5370 async fn busy_work_graph_still_runs_the_command() {
5371 use crate::tools::plan::new_shared_plan_state;
5372 use crate::tools::todo::new_shared_todo_list;
5373 use crate::work_graph::new_shared_work_runtime;
5374
5375 let workspace = tempdir().expect("workspace");
5376 let todos = new_shared_todo_list();
5377 let plan = new_shared_plan_state();
5378 let lifecycle = ShellWorkLifecycle {
5379 work: new_shared_work_runtime(todos.clone(), plan.clone()),
5380 session_id: "session-test".to_string(),
5381 };
5382 let _held = todos.lock().await;
5383 let mut manager = ShellManager::new(workspace.path().to_path_buf());
5384 for background in [false, true] {
5385 let marker = format!("ran-{background}.txt");
5386 let result = manager
5387 .execute_with_options_env_for_owner_and_work(
5388 &format!("echo ran > {marker}"),
5389 None,
5390 10_000,
5391 background,
5392 None,
5393 false,
5394 None,
5395 HashMap::new(),
5396 None,
5397 "session-test".to_string(),
5398 None,
5399 None,
5400 Some(lifecycle.clone()),
5401 None,
5402 false,
5403 (1_000, 60_000),
5404 )
5405 .unwrap_or_else(|err| panic!("background={background}: {err:#}"));
5406 if background {
5407 let task_id = result.task_id.expect("background task id");
5408 let deadline = std::time::Instant::now() + Duration::from_secs(10);
5409 while !workspace.path().join(&marker).exists() {
5410 assert!(std::time::Instant::now() < deadline, "{task_id} never ran");
5411 tokio::time::sleep(Duration::from_millis(10)).await;
5412 }
5413 }
5414 assert!(
5415 workspace.path().join(&marker).exists(),
5416 "background={background}: the command ran"
5417 );
5418 }
5419 }
5420
5421 #[test]
5422 fn pty_dimensions_reject_zero_and_unbounded_grid() {
5423 assert_eq!(
5424 PtyDimensions::default(),
5425 PtyDimensions { rows: 24, cols: 80 }
5426 );
5427 for size in [
5428 PtyDimensions { rows: 0, cols: 80 },
5429 PtyDimensions { rows: 24, cols: 0 },
5430 PtyDimensions {
5431 rows: 1001,
5432 cols: 80,
5433 },
5434 PtyDimensions {
5435 rows: 24,
5436 cols: u16::MAX,
5437 },
5438 ] {
5439 assert!(size.validate().is_err());
5440 }
5441 assert!(
5442 PtyDimensions {
5443 rows: 1000,
5444 cols: 1000
5445 }
5446 .validate()
5447 .is_ok()
5448 );
5449 }
5450
5451 #[test]
5452 #[cfg(not(target_env = "ohos"))]
5453 fn pty_stdin_preserves_bytes_and_reports_flush_failure() {
5454 struct FlushFailure(Arc<Mutex<Vec<u8>>>);
5455 impl Write for FlushFailure {
5456 fn write(&mut self, bytes: &[u8]) -> io::Result<usize> {
5457 self.0.lock().unwrap().extend_from_slice(bytes);
5458 Ok(bytes.len())
5459 }
5460 fn flush(&mut self) -> io::Result<()> {
5461 Err(io::Error::new(
5462 io::ErrorKind::BrokenPipe,
5463 "fixture flush failure",
5464 ))
5465 }
5466 }
5467 let tmp = tempdir().unwrap();
5468 let mut manager = ShellManager::new(tmp.path().to_path_buf());
5469 manager.seed_finished_record_for_test("input-fixture", Duration::ZERO);
5470 let bytes = Arc::new(Mutex::new(Vec::new()));
5471 manager.processes.get_mut("input-fixture").unwrap().stdin =
5472 Some(StdinWriter::Pty(Box::new(FlushFailure(bytes.clone()))));
5473 let input = b"\0\xff\x1b[A\x03";
5474 let error = manager
5475 .write_stdin_bytes("input-fixture", input, false)
5476 .unwrap_err();
5477 assert!(error.to_string().contains("flush"));
5478 assert_eq!(bytes.lock().unwrap().as_slice(), input);
5479 }
5480
5481 #[test]
5482 #[cfg(all(unix, not(target_env = "ohos")))]
5483 fn pty_resize_updates_live_terminal_and_rejects_finished_or_pipe_jobs() {
5484 let tmp = tempdir().unwrap();
5485 let mut manager = ShellManager::new(tmp.path().to_path_buf());
5486 let launched = manager
5487 .execute_with_options_env(
5488 "stty -echo; printf ready; while IFS= read -r line; do stty size; done",
5489 None,
5490 5000,
5491 true,
5492 None,
5493 true,
5494 Some(ExecutionSandboxPolicy::DangerFullAccess),
5495 HashMap::new(),
5496 )
5497 .unwrap();
5498 let id = launched.task_id.unwrap();
5499 assert_eq!(
5500 manager.job_terminal_size(&id),
5501 Some(PtyDimensions::default())
5502 );
5503 let deadline = Instant::now() + Duration::from_secs(5);
5504 loop {
5505 let chunk = manager
5506 .read_output_chunk(&id, ShellOutputStream::Stdout, 0, 4096, 50)
5507 .unwrap();
5508 if chunk.bytes.windows(5).any(|bytes| bytes == b"ready") {
5509 break;
5510 }
5511 assert!(Instant::now() < deadline, "PTY did not become ready");
5512 }
5513 // Reopening a client after an idle minute must retain this live PTY's
5514 // identity and resize eligibility; output silence is normal at a prompt.
5515 manager.processes.get_mut(&id).unwrap().last_output_at =
5516 Instant::now() - STALE_NO_OUTPUT_AFTER - Duration::from_millis(1);
5517 let idle = manager.inspect_job(&id).unwrap().snapshot;
5518 assert_eq!(idle.status, ShellStatus::Running);
5519 assert!(idle.stdin_available);
5520 assert!(!idle.stale, "an idle live PTY must remain reconnectable");
5521 assert!(
5522 idle.elapsed_since_output_ms
5523 .is_some_and(|ms| ms >= STALE_NO_OUTPUT_AFTER.as_millis() as u64)
5524 );
5525 let size = PtyDimensions {
5526 rows: 37,
5527 cols: 111,
5528 };
5529 manager.resize_pty(&id, size).unwrap();
5530 assert_eq!(manager.job_terminal_size(&id), Some(size));
5531 assert!(
5532 manager
5533 .resize_pty(&id, PtyDimensions { rows: 0, cols: 1 })
5534 .is_err()
5535 );
5536 assert_eq!(manager.job_terminal_size(&id), Some(size));
5537 manager.write_stdin_bytes(&id, b"size\n", false).unwrap();
5538 loop {
5539 let chunk = manager
5540 .read_output_chunk(&id, ShellOutputStream::Stdout, 0, 4096, 0)
5541 .unwrap();
5542 if String::from_utf8_lossy(&chunk.bytes).contains("37 111") {
5543 break;
5544 }
5545 assert!(
5546 Instant::now() < deadline,
5547 "actual terminal size did not change"
5548 );
5549 std::thread::sleep(Duration::from_millis(20));
5550 }
5551 manager.kill(&id).unwrap();
5552 assert!(manager.resize_pty(&id, size).is_err());
5553 assert!(manager.processes[&id].pty_master.is_none());
5554 let pipe = manager
5555 .execute_with_options_env(
5556 "cat",
5557 None,
5558 5000,
5559 true,
5560 None,
5561 false,
5562 Some(ExecutionSandboxPolicy::DangerFullAccess),
5563 HashMap::new(),
5564 )
5565 .unwrap()
5566 .task_id
5567 .unwrap();
5568 assert!(manager.resize_pty(&pipe, size).is_err());
5569 manager.kill(&pipe).unwrap();
5570 }
5571
5572 #[test]
5573 #[cfg(all(unix, not(target_env = "ohos")))]
5574 fn pty_raw_stdin_roundtrips_nul_and_non_utf8_bytes() {
5575 let tmp = tempdir().unwrap();
5576 let mut manager = ShellManager::new(tmp.path().to_path_buf());
5577 let input = b"\0\xff\x1b[A\x03\n";
5578 let command = format!(
5579 "stty raw -echo; printf ready; dd bs=1 count={} 2>/dev/null",
5580 input.len()
5581 );
5582 let launched = manager
5583 .execute_with_options_env(
5584 &command,
5585 None,
5586 5000,
5587 true,
5588 None,
5589 true,
5590 Some(ExecutionSandboxPolicy::DangerFullAccess),
5591 HashMap::new(),
5592 )
5593 .unwrap();
5594 let id = launched.task_id.unwrap();
5595 let deadline = Instant::now() + Duration::from_secs(5);
5596 loop {
5597 let chunk = manager
5598 .read_output_chunk(&id, ShellOutputStream::Stdout, 0, 4096, 0)
5599 .unwrap();
5600 if chunk.bytes == b"ready" {
5601 break;
5602 }
5603 assert!(Instant::now() < deadline, "raw PTY did not become ready");
5604 std::thread::sleep(Duration::from_millis(20));
5605 }
5606 manager.write_stdin_bytes(&id, input, false).unwrap();
5607 loop {
5608 let chunk = manager
5609 .read_output_chunk(&id, ShellOutputStream::Stdout, 5, 4096, 0)
5610 .unwrap();
5611 if chunk.status != ShellStatus::Running {
5612 assert_eq!(chunk.bytes, input);
5613 assert_eq!(chunk.next_offset, 5 + input.len());
5614 break;
5615 }
5616 assert!(Instant::now() < deadline, "raw PTY did not finish");
5617 std::thread::sleep(Duration::from_millis(20));
5618 }
5619 }
5620
5621 /// #6654: a `Managed` background shell must (a) survive the exit of the thread
5622 /// that spawned it — callers such as `spawn_blocking` retire their threads —
5623 /// and (b) die with its whole process group when the TUI process is
5624 /// SIGKILLed, including a grandchild the shell did not `exec` (the shell
5625 /// itself keeps waiting on it). The TUI is played by a re-exec of this test
5626 /// binary.
5627 #[cfg(unix)]
5628 #[test]
5629 fn managed_background_shell_outlives_spawning_thread_but_not_the_tui() {
5630 const PROBE: &str = "CODEWHALE_TEST_BG_PDEATHSIG_PIDFILE";
5631
5632 #[cfg(target_os = "linux")]
5633 fn is_alive(pid: libc::pid_t) -> bool {
5634 // A killed child may linger as a zombie until its new parent reaps it
5635 // (a container's init may never do so).
5636 match std::fs::read_to_string(format!("/proc/{pid}/stat")) {
5637 Ok(stat) => stat
5638 .rsplit_once(')')
5639 .and_then(|(_, rest)| rest.split_whitespace().next())
5640 .is_some_and(|state| state != "Z" && state != "X"),
5641 Err(_) => false,
5642 }
5643 }
5644
5645 #[cfg(not(target_os = "linux"))]
5646 fn is_alive(pid: libc::pid_t) -> bool {
5647 // Orphans are reparented to launchd, which reaps them promptly.
5648 // SAFETY: kill(2) with signal 0 only checks for existence.
5649 unsafe { libc::kill(pid, 0) == 0 }
5650 }
5651
5652 if let Some(pid_file) = std::env::var_os(PROBE) {
5653 // Probe role: spawn from a short-lived thread, let the thread exit,
5654 // mark readiness (libtest captures stdout, so through a file), then
5655 // hang until SIGKILLed.
5656 let pid_file = PathBuf::from(pid_file);
5657 let workspace = pid_file.parent().unwrap().to_path_buf();
5658 // No `exec`: the shell forks `sleep` and waits, as `npm run dev; echo
5659 // done` or `a | b` would.
5660 let command = format!(
5661 "sleep 600 & echo \"$$ $!\" > '{}'; wait",
5662 pid_file.display()
5663 );
5664 let manager = std::thread::spawn(move || {
5665 let mut manager = ShellManager::new(workspace);
5666 let result = manager
5667 .execute_with_options_env(
5668 &command,
5669 None,
5670 600_000,
5671 true,
5672 None,
5673 false,
5674 Some(ExecutionSandboxPolicy::DangerFullAccess),
5675 HashMap::new(),
5676 )
5677 .expect("spawn background shell");
5678 assert_eq!(result.status, ShellStatus::Running);
5679 manager
5680 })
5681 .join()
5682 .expect("spawning thread");
5683 std::fs::write(pid_file.with_extension("ready"), b"").expect("write ready marker");
5684 let _keep_alive = manager;
5685 loop {
5686 std::thread::sleep(Duration::from_secs(60));
5687 }
5688 }
5689
5690 let tmp = tempdir().expect("tempdir");
5691 let pid_file = tmp.path().join("bg.pid");
5692 let ready_file = pid_file.with_extension("ready");
5693 let mut tui = std::process::Command::new(std::env::current_exe().unwrap())
5694 .args([
5695 "--exact",
5696 "tools::shell::tests::managed_background_shell_outlives_spawning_thread_but_not_the_tui",
5697 "--test-threads=1",
5698 ])
5699 .env(PROBE, &pid_file)
5700 .stdout(Stdio::null())
5701 .stderr(Stdio::null())
5702 .spawn()
5703 .expect("spawn probe process");
5704 let deadline = Instant::now() + Duration::from_secs(30);
5705 let pids = loop {
5706 let pids = if ready_file.exists() {
5707 std::fs::read_to_string(&pid_file).ok().and_then(|text| {
5708 let pids = text
5709 .split_whitespace()
5710 .map(|pid| pid.parse::<libc::pid_t>().ok())
5711 .collect::<Option<Vec<_>>>()?;
5712 (pids.len() == 2).then_some(pids)
5713 })
5714 } else {
5715 None
5716 };
5717 if let Some(pids) = pids {
5718 break pids;
5719 }
5720 if Instant::now() >= deadline || tui.try_wait().ok().flatten().is_some() {
5721 let _ = tui.kill();
5722 let _ = tui.wait();
5723 panic!("probe process never reported the background shell pid");
5724 }
5725 std::thread::sleep(Duration::from_millis(20));
5726 };
5727
5728 // The spawning thread is gone by now; give a mis-armed signal time to land.
5729 std::thread::sleep(Duration::from_millis(500));
5730 let survived_thread_exit = pids.iter().all(|&pid| is_alive(pid));
5731
5732 tui.kill().expect("SIGKILL the probe process");
5733 tui.wait().expect("reap the probe process");
5734 let deadline = Instant::now() + Duration::from_secs(10);
5735 while pids.iter().any(|&pid| is_alive(pid)) && Instant::now() < deadline {
5736 std::thread::sleep(Duration::from_millis(20));
5737 }
5738 let survivors = pids
5739 .iter()
5740 .copied()
5741 .filter(|&pid| is_alive(pid))
5742 .collect::<Vec<_>>();
5743 for &pid in &survivors {
5744 // SAFETY: plain kill(2) on a pid this test started.
5745 unsafe {
5746 libc::kill(pid, libc::SIGKILL);
5747 }
5748 }
5749
5750 assert!(
5751 survived_thread_exit,
5752 "the background shell died when its spawning thread exited"
5753 );
5754 assert!(
5755 survivors.is_empty(),
5756 "background processes outlived the SIGKILLed TUI: {survivors:?} of shell/grandchild {pids:?}"
5757 );
5758 }
5759
5760 /// The auto-approved `note` tool appends to the configured notes file. A
5761 /// committed symlink (`notes.md -> ~/.zshrc`) or a symlinked notes directory
5762 /// must not redirect that append outside the workspace.
5763 #[cfg(unix)]
5764 #[tokio::test]
5765 async fn note_tool_refuses_symlinked_targets_that_leave_the_workspace() {
5766 let workspace = tempdir().expect("workspace");
5767 let outside = tempdir().expect("outside");
5768 let rc = outside.path().join(".zshrc");
5769 std::fs::write(&rc, "# rc\n").expect("write rc");
5770
5771 let linked_file = workspace.path().join("notes.md");
5772 std::os::unix::fs::symlink(&rc, &linked_file).expect("symlink notes file");
5773 let context = ToolContext::with_options(workspace.path(), false, &linked_file, "mcp.json");
5774 let err = NoteTool
5775 .execute(json!({"content": "appended text"}), &context)
5776 .await
5777 .expect_err("a symlinked notes file must be refused");
5778 assert!(err.to_string().contains("symlink"), "{err}");
5779
5780 let linked_dir = workspace.path().join("notes");
5781 std::os::unix::fs::symlink(outside.path(), &linked_dir).expect("symlink notes dir");
5782 let context = ToolContext::with_options(
5783 workspace.path(),
5784 false,
5785 linked_dir.join("sub/notes.md"),
5786 "mcp.json",
5787 );
5788 let err = NoteTool
5789 .execute(json!({"content": "appended text"}), &context)
5790 .await
5791 .expect_err("a notes dir resolving outside the workspace must be refused");
5792 assert!(err.to_string().contains("outside the workspace"), "{err}");
5793
5794 assert_eq!(std::fs::read_to_string(&rc).expect("read rc"), "# rc\n");
5795 assert!(!outside.path().join("sub").exists());
5796
5797 let plain = workspace.path().join("docs/notes.md");
5798 let context = ToolContext::with_options(workspace.path(), false, &plain, "mcp.json");
5799 NoteTool
5800 .execute(json!({"content": "kept"}), &context)
5801 .await
5802 .expect("an ordinary in-workspace notes file is writable");
5803 assert!(
5804 std::fs::read_to_string(&plain)
5805 .expect("read notes")
5806 .contains("kept")
5807 );
5808 }
5809
5809 lines RUST