返回 CodeWhale
test_runner.rs
根目录 / crates / tui / src / tools / test_runner.rs
1 //! Cargo test runner tool: `run_tests`.
2 //!
3 //! `cargo test` runs workspace code, so this tool follows the same explicit
4 //! approval policy as the other code-executing tools.
5
6 use std::path::Path;
7 use std::time::Duration;
8
9 use async_trait::async_trait;
10 use serde::{Deserialize, Serialize};
11 use serde_json::{Value, json};
12
13 use super::cargo_failure_summary::summarize_cargo_failure;
14 use super::spec::{
15 ApprovalRequirement, ToolCapability, ToolContext, ToolError, ToolResult, ToolSpec,
16 optional_bool, optional_str,
17 };
18
19 use crate::dependencies::ExternalTool;
20
21 /// Tool for running `cargo test` in the workspace root.
22 pub struct RunTestsTool;
23
24 #[derive(Debug, Clone, Serialize, Deserialize)]
25 struct RunTestsOutput {
26 success: bool,
27 exit_code: i32,
28 stdout: String,
29 stderr: String,
30 command: String,
31 /// The run hit [`RUN_TESTS_TIMEOUT`] and was killed; stdout/stderr are
32 /// what it wrote until then.
33 #[serde(default, skip_serializing_if = "std::ops::Not::not")]
34 timed_out: bool,
35 }
36
37 #[async_trait]
38 impl ToolSpec for RunTestsTool {
39 fn name(&self) -> &'static str {
40 "run_tests"
41 }
42
43 fn model_visible(&self) -> bool {
44 false
45 }
46
47 fn description(&self) -> &'static str {
48 "Run `cargo test` in the workspace root with optional extra arguments."
49 }
50
51 fn input_schema(&self) -> Value {
52 json!({
53 "type": "object",
54 "properties": {
55 "args": {
56 "type": "string",
57 "description": "Optional extra arguments to pass to `cargo test` (shell-style)."
58 },
59 "all_features": {
60 "type": "boolean",
61 "description": "When true, include `--all-features`."
62 },
63 "cwd": {
64 "type": "string",
65 "description": "Optional working directory, relative to the workspace, to run `cargo test` in. Must exist inside the workspace."
66 }
67 },
68 "additionalProperties": false
69 })
70 }
71
72 fn capabilities(&self) -> Vec<ToolCapability> {
73 vec![ToolCapability::ExecutesCode, ToolCapability::Sandboxable]
74 }
75
76 fn approval_requirement(&self) -> ApprovalRequirement {
77 // `run_tests` declares `ToolCapability::ExecutesCode` — match the
78 // default approval policy for code-executing tools.
79 ApprovalRequirement::Required
80 }
81
82 async fn execute(&self, input: Value, context: &ToolContext) -> Result<ToolResult, ToolError> {
83 crate::core::engine::tool_catalog::enforce_tool_denial(context, self.name(), &input)?;
84 let all_features = optional_bool(&input, "all_features", false)?;
85 let extra_args = optional_str(&input, "args")?
86 .map(str::trim)
87 .filter(|s| !s.is_empty());
88 let workdir = match optional_str(&input, "cwd")?
89 .map(str::trim)
90 .filter(|s| !s.is_empty())
91 {
92 None => context.workspace.clone(),
93 Some(raw) => context.resolve_existing_dir(raw, "cwd")?,
94 };
95
96 let mut args = vec!["test".to_string()];
97 if all_features {
98 args.push("--all-features".to_string());
99 }
100 if let Some(extra) = extra_args {
101 let split = shlex::split(extra).ok_or_else(|| {
102 ToolError::invalid_input("Failed to parse 'args' as shell-style tokens")
103 })?;
104 args.extend(split);
105 }
106
107 let command_str = format_command(&workdir, &args);
108 let run = run_cargo(&workdir, &args, context, RUN_TESTS_TIMEOUT).await?;
109 let output = run.output;
110 let mut stderr = String::from_utf8_lossy(&output.stderr).into_owned();
111 if run.stopped {
112 stderr.push_str(&format!(
113 "\n[run_tests: stopped after {} s; the output above is everything cargo wrote before it was killed]",
114 RUN_TESTS_TIMEOUT.as_secs()
115 ));
116 }
117
118 // The whole output: the end of a cargo run is where the failures and
119 // the `test result:` line are. Size is the engine's one recoverable
120 // budget (#6508), so the failure summary sees everything.
121 run_tests_result(RunTestsOutput {
122 success: !run.stopped && output.status.success(),
123 exit_code: output.status.code().unwrap_or(-1),
124 stdout: String::from_utf8_lossy(&output.stdout).into_owned(),
125 stderr,
126 command: command_str,
127 timed_out: run.stopped,
128 })
129 }
130 }
131
132 fn run_tests_result(result: RunTestsOutput) -> Result<ToolResult, ToolError> {
133 let mut tool_result =
134 ToolResult::json(&result).map_err(|e| ToolError::execution_failed(e.to_string()))?;
135 if let Some(summary) = summarize_cargo_failure(
136 &result.command,
137 &result.stdout,
138 &result.stderr,
139 Some(result.exit_code),
140 ) {
141 tool_result = tool_result.with_metadata(json!({
142 "summary": summary.summary,
143 "cargo_failure_summary": summary.to_metadata_value(),
144 }));
145 }
146 Ok(tool_result)
147 }
148
149 // === Helpers ===
150
151 /// Ceiling for one `cargo test` run. Long suites fit; a hung test does not
152 /// hold the turn forever.
153 const RUN_TESTS_TIMEOUT: Duration = Duration::from_secs(30 * 60);
154
155 /// Run cargo without blocking a runtime worker. Stop and the timeout both end
156 /// the whole process tree (cargo and the test binaries it started). A timeout
157 /// still returns the output written so far (`stopped`): it is the only
158 /// evidence of which test hung.
159 async fn run_cargo(
160 workspace: &Path,
161 args: &[String],
162 context: &ToolContext,
163 timeout: Duration,
164 ) -> Result<crate::process_tree::ContainedOutput, ToolError> {
165 let Some(cargo) = crate::dependencies::Cargo::resolve() else {
166 return Err(ToolError::not_available(
167 "cargo is not installed or not in PATH",
168 ));
169 };
170 // `cargo test` builds and runs workspace code: it starts like an
171 // `exec_shell` command, inside this session's sandbox and without parent
172 // credentials.
173 let mut cmd = crate::tools::shell::sandboxed_runner_command(
174 context,
175 &cargo,
176 args.to_vec(),
177 workspace,
178 timeout,
179 )?;
180 let run = crate::process_tree::contained_output_until(&mut cmd, tokio::time::sleep(timeout));
181 let cancelled = async {
182 match context.cancel_token.as_ref() {
183 Some(token) => token.cancelled().await,
184 None => std::future::pending::<()>().await,
185 }
186 };
187 // Dropping `run` on cancel kills the tree.
188 let output = tokio::select! {
189 output = run => output,
190 () = cancelled => return Err(ToolError::cancelled("cargo test cancelled")),
191 };
192 output.map_err(|e| {
193 if e.kind() == std::io::ErrorKind::NotFound {
194 ToolError::not_available("cargo is not installed or not in PATH")
195 } else {
196 ToolError::execution_failed(format!("Failed to run cargo: {e}"))
197 }
198 })
199 }
200
201 fn format_command(workspace: &Path, args: &[String]) -> String {
202 format!(
203 "(cd {} && cargo {})",
204 workspace.display(),
205 args.iter()
206 .map(String::as_str)
207 .collect::<Vec<_>>()
208 .join(" ")
209 )
210 }
211
212 #[cfg(test)]
213 mod tests {
214 use super::*;
215 use std::fs;
216 use std::process::Command;
217 use std::sync::atomic::{AtomicU64, Ordering};
218 use tempfile::tempdir;
219
220 static NEXT_CARGO_PROJECT: AtomicU64 = AtomicU64::new(0);
221
222 fn cargo_available() -> bool {
223 Command::new("cargo")
224 .arg("--version")
225 .output()
226 .map(|o| o.status.success())
227 .unwrap_or(false)
228 }
229
230 fn init_cargo_project(root: &Path) -> std::path::PathBuf {
231 let project_dir = root.join("project");
232 let package_name = format!(
233 "eval_project_{}_{}",
234 std::process::id(),
235 NEXT_CARGO_PROJECT.fetch_add(1, Ordering::Relaxed)
236 );
237 fs::create_dir_all(&project_dir).expect("create project dir");
238 let status = crate::dependencies::Cargo::command()
239 .expect("cargo not found")
240 .args(["init", "--lib", "--vcs", "none", "-q"])
241 .arg("--name")
242 .arg(package_name)
243 .current_dir(&project_dir)
244 .status()
245 .expect("cargo should spawn");
246 assert!(status.success(), "cargo init failed");
247 project_dir
248 }
249
250 /// `run_tests` is `ToolCapability::ExecutesCode`, so it must follow the
251 /// explicit-approval policy that applies to other code-executing tools.
252 #[test]
253 fn run_tests_requires_user_approval() {
254 let tool = RunTestsTool;
255 assert_eq!(
256 tool.approval_requirement(),
257 ApprovalRequirement::Required,
258 "run_tests must gate cargo test behind user approval"
259 );
260 }
261
262 /// Stop must reach `cargo test`: the call returns as cancelled instead of
263 /// blocking a runtime worker until cargo exits on its own.
264 #[tokio::test]
265 async fn run_tests_honors_cancellation() {
266 if !cargo_available() {
267 return;
268 }
269 let tmp = tempdir().expect("tempdir");
270 let token = tokio_util::sync::CancellationToken::new();
271 token.cancel();
272 let ctx = ToolContext::new(tmp.path()).with_cancel_token(token);
273 let result = RunTestsTool.execute(json!({}), &ctx).await;
274 assert!(
275 matches!(result, Err(ToolError::Cancelled { .. })),
276 "cancelled run_tests must not run to completion: {result:?}"
277 );
278 }
279
280 #[tokio::test]
281 async fn run_tests_succeeds_on_fresh_project() {
282 if !cargo_available() {
283 return;
284 }
285 let tmp = tempdir().expect("tempdir");
286 // Release jobs commonly export one CARGO_TARGET_DIR for the whole
287 // workspace. Give concurrent nested Cargo fixtures distinct package
288 // identities so their test artifacts cannot replace each other.
289 let project_dir = init_cargo_project(tmp.path());
290
291 let ctx = ToolContext::new(&project_dir);
292 let tool = RunTestsTool;
293 let result = tool.execute(json!({}), &ctx).await.expect("execute");
294 assert!(result.success);
295
296 let parsed: RunTestsOutput =
297 serde_json::from_str(&result.content).expect("tool result should be json");
298 assert!(
299 parsed.success,
300 "nested cargo test unexpectedly failed:\n{}",
301 parsed.stderr
302 );
303 assert_eq!(parsed.exit_code, 0);
304 assert!(parsed.command.contains("cargo test"));
305 }
306
307 #[tokio::test]
308 async fn run_tests_reports_failures_without_hard_error() {
309 if !cargo_available() {
310 return;
311 }
312 let tmp = tempdir().expect("tempdir");
313 let project_dir = init_cargo_project(tmp.path());
314
315 let lib_rs = project_dir.join("src/lib.rs");
316 let failing = r#"
317 pub fn add(a: i32, b: i32) -> i32 { a + b }
318
319 #[cfg(test)]
320 mod tests {
321 #[test]
322 fn fails() {
323 assert_eq!(2 + 2, 5);
324 }
325 }
326 "#;
327 fs::write(&lib_rs, failing).expect("write failing test");
328
329 let ctx = ToolContext::new(&project_dir);
330 let tool = RunTestsTool;
331 let result = tool.execute(json!({}), &ctx).await.expect("execute");
332 assert!(result.success);
333
334 let parsed: RunTestsOutput =
335 serde_json::from_str(&result.content).expect("tool result should be json");
336 assert!(
337 !parsed.success,
338 "nested cargo test unexpectedly passed:\nstdout:\n{}\nstderr:\n{}",
339 parsed.stdout, parsed.stderr
340 );
341 assert_ne!(parsed.exit_code, 0);
342 let metadata = result.metadata.expect("metadata");
343 assert_eq!(
344 metadata["cargo_failure_summary"]["kind"],
345 json!("test_failure")
346 );
347 assert!(
348 metadata["cargo_failure_summary"]["summary"]
349 .as_str()
350 .unwrap()
351 .contains("Failing tests:")
352 );
353 }
354
355 #[test]
356 fn long_failing_output_comes_back_whole_and_the_summary_sees_its_end() {
357 // #6508: stdout used to be cut at 40,000 characters, so the model
358 // read passing tests and never saw which one failed, and the failure
359 // summary ran on the truncated text.
360 let stdout = format!(
361 "{}test tools::git::tests::diff_keeps_the_last_file ... FAILED\n\ntest result: FAILED. 3000 passed; 1 failed; 0 ignored",
362 "test tools::ok ... ok\n".repeat(3_000)
363 );
364 assert!(stdout.chars().count() > 40_000);
365 let result = run_tests_result(RunTestsOutput {
366 success: false,
367 exit_code: 101,
368 stdout: stdout.clone(),
369 stderr: String::new(),
370 command: "(cd /repo && cargo test)".to_string(),
371 timed_out: false,
372 })
373 .expect("result");
374
375 let parsed: Value = serde_json::from_str(&result.content).expect("json");
376 assert_eq!(parsed["stdout"], json!(stdout));
377 let summary = result
378 .metadata
379 .as_ref()
380 .and_then(|metadata| metadata.get("summary"))
381 .and_then(Value::as_str)
382 .expect("failure summary");
383 assert!(
384 summary.contains("diff_keeps_the_last_file"),
385 "summary: {summary}"
386 );
387 }
388
389 /// A child parked at a workspace root that is not the project root (the
390 /// #6296 verifier) runs the suite where the manifest lives instead of
391 /// failing on cwd.
392 #[tokio::test]
393 async fn run_tests_cwd_scopes_cargo_to_subdir() {
394 if !cargo_available() {
395 return;
396 }
397 let tmp = tempdir().expect("tempdir");
398 let project_dir = init_cargo_project(tmp.path());
399
400 let ctx = ToolContext::new(tmp.path());
401 let result = RunTestsTool
402 .execute(json!({"cwd": "project"}), &ctx)
403 .await
404 .expect("cwd-scoped execute");
405 assert!(result.success);
406
407 let parsed: RunTestsOutput =
408 serde_json::from_str(&result.content).expect("tool result should be json");
409 assert!(
410 parsed.success,
411 "nested cargo test unexpectedly failed:\\n{}",
412 parsed.stderr
413 );
414 // `resolve_existing_dir` returns the canonical path, which on Windows
415 // carries the `\\?\` verbatim prefix the raw tempdir lacks (#6346).
416 let scoped_dir = project_dir.canonicalize().expect("canonical project dir");
417 assert!(
418 parsed.command.contains(&scoped_dir.display().to_string()),
419 "cargo must run in the scoped dir, ran: {}",
420 parsed.command
421 );
422 }
423
424 #[tokio::test]
425 async fn run_tests_cwd_fails_closed_with_a_named_fallback() {
426 let tmp = tempdir().expect("tempdir");
427 let ctx = ToolContext::new(tmp.path());
428
429 let escape = RunTestsTool
430 .execute(json!({"cwd": "../escape"}), &ctx)
431 .await
432 .expect_err("workspace escape must be refused");
433 assert!(escape.to_string().contains("escapes workspace"), "{escape}");
434
435 let missing = RunTestsTool
436 .execute(json!({"cwd": "no-such-dir"}), &ctx)
437 .await
438 .expect_err("missing dir must be refused");
439 let message = missing.to_string();
440 assert!(message.contains("not an existing directory"), "{message}");
441 assert!(message.contains("drop `cwd`"), "{message}");
442 }
443
444 #[cfg(unix)]
445 #[tokio::test]
446 async fn run_cargo_does_not_inherit_parent_secret_env() {
447 use crate::test_support::{EnvVarGuard, lock_test_env};
448 use std::os::unix::fs::PermissionsExt;
449 if !cargo_available() {
450 return;
451 }
452 let _env_lock = lock_test_env();
453 let bin = tempdir().expect("bin dir");
454 // cargo resolves `cargo envprobe` to a `cargo-envprobe` on PATH, which
455 // lets the test observe the environment cargo hands its children.
456 let probe = bin.path().join("cargo-envprobe");
457 fs::write(
458 &probe,
459 "#!/bin/sh\nprintf 'secret=%s target=%s' \"${CODEWHALE_TEST_CARGO_SECRET-unset}\" \"${CARGO_TARGET_DIR-unset}\"\n",
460 )
461 .expect("write probe");
462 fs::set_permissions(&probe, fs::Permissions::from_mode(0o755)).expect("chmod probe");
463 let path = std::env::var_os("PATH").unwrap_or_default();
464 let mut paths = vec![bin.path().to_path_buf()];
465 paths.extend(std::env::split_paths(&path));
466 let _path = EnvVarGuard::set("PATH", std::env::join_paths(paths).expect("join PATH"));
467 let _secret = EnvVarGuard::set("CODEWHALE_TEST_CARGO_SECRET", "cargo-secret-value");
468 // Non-secret build configuration still reaches cargo.
469 let _target = EnvVarGuard::set("CARGO_TARGET_DIR", "/tmp/codewhale-fixture-target");
470 let workspace = tempdir().expect("workspace");
471
472 let ctx = ToolContext::new(workspace.path());
473 let run = run_cargo(
474 workspace.path(),
475 &["envprobe".to_string()],
476 &ctx,
477 RUN_TESTS_TIMEOUT,
478 )
479 .await
480 .expect("cargo runs");
481 assert!(!run.stopped);
482 let output = run.output;
483 let stdout = String::from_utf8_lossy(&output.stdout);
484 assert!(
485 output.status.success(),
486 "{stdout} {}",
487 String::from_utf8_lossy(&output.stderr)
488 );
489 assert_eq!(
490 stdout.trim(),
491 "secret=unset target=/tmp/codewhale-fixture-target"
492 );
493 }
494
495 /// A run that hits the timeout is killed, tree and all, and still returns
496 /// what it wrote: the only evidence of which test hung.
497 #[cfg(unix)]
498 #[tokio::test]
499 async fn timed_out_run_cargo_keeps_partial_output_and_kills_the_tree() {
500 use crate::test_support::{EnvVarGuard, lock_test_env};
501 use std::os::unix::fs::PermissionsExt;
502 if !cargo_available() {
503 return;
504 }
505 let _env_lock = lock_test_env();
506 let bin = tempdir().expect("bin dir");
507 let probe = bin.path().join("cargo-hangprobe");
508 fs::write(
509 &probe,
510 "#!/bin/sh\necho 'test slow_case ... started'\nsleep 300 &\necho $! > hang.pid\nwait\n",
511 )
512 .expect("write probe");
513 fs::set_permissions(&probe, fs::Permissions::from_mode(0o755)).expect("chmod probe");
514 let path = std::env::var_os("PATH").unwrap_or_default();
515 let mut paths = vec![bin.path().to_path_buf()];
516 paths.extend(std::env::split_paths(&path));
517 let _path = EnvVarGuard::set("PATH", std::env::join_paths(paths).expect("join PATH"));
518 let workspace = tempdir().expect("workspace");
519 let pid_file = workspace.path().join("hang.pid");
520
521 let ctx = ToolContext::new(workspace.path());
522 let run = run_cargo(
523 workspace.path(),
524 &["hangprobe".to_string()],
525 &ctx,
526 Duration::from_secs(10),
527 )
528 .await
529 .expect("a timed-out run still returns its output");
530 assert!(run.stopped);
531 assert!(
532 String::from_utf8_lossy(&run.output.stdout).contains("test slow_case ... started"),
533 "{:?}",
534 run.output
535 );
536 let hung = crate::process_tree::read_pid_file(&pid_file, Duration::from_secs(5));
537 assert!(
538 crate::process_tree::wait_for_pid_exit(hung, Duration::from_secs(5)),
539 "the timed-out run's process is still running"
540 );
541 }
542 }
543
543 lines RUST