返回 CodeWhale
eval_smoke_acceptance.rs
根目录 / crates / tui / tests / cucumber / eval_smoke_acceptance.rs
1 //! Gherkin acceptance test: eval smoke test.
2 //!
3 //! Verifies that the binary loads and the eval harness reports step-level
4 //! success for a shell command after Layer 4.2 registry cleanup. Follows the
5 //! proven `core_session_command_extraction.rs` pattern.
6 //!
7 //! NOTE: This is an eval smoke test, not a command-surface verification
8 //! (AT-004) test. It confirms the binary starts and runs eval correctly.
9 //! For AT-004 command-surface coverage (help, palette, completion), see the
10 //! focused unit tests in command_palette.rs, widgets/mod.rs, and
11 //! commands/mod.rs.
12
13 use std::process::{Command, ExitStatus};
14
15 use cucumber::{World as _, given, then, when, writer::Stats as _};
16 use serde_json::Value;
17 use tempfile::TempDir;
18
19 const FEATURE_NAME: &str = "Eval smoke test (binary load and eval step reporting)";
20 const FEATURE_PATH: &str = concat!(
21 env!("CARGO_MANIFEST_DIR"),
22 "/tests/features/eval_smoke.feature"
23 );
24 const SMOKE_SCENARIO: &str = "Binary loads and reports step-level success via eval";
25
26 #[derive(Debug, Default, cucumber::World)]
27 struct EvalSmokeWorld {
28 _record_dir: Option<TempDir>,
29 report: Option<Value>,
30 exit_status: Option<ExitStatus>,
31 }
32
33 #[given("a clean CodeWhale evaluation workspace")]
34 fn clean_codewhale_evaluation_workspace(world: &mut EvalSmokeWorld) {
35 world._record_dir = Some(TempDir::new().expect("evaluation TempDir"));
36 }
37
38 #[when("the evaluation harness runs a shell command")]
39 fn eval_harness_runs_shell_command(world: &mut EvalSmokeWorld) {
40 let record_dir = world
41 ._record_dir
42 .as_ref()
43 .expect("evaluation workspace should exist");
44
45 let output = Command::new(crate::binary::codewhale())
46 .args([
47 "eval",
48 "--json",
49 "--shell-command",
50 "echo eval-smoke-test",
51 "--record",
52 ])
53 .arg(record_dir.path())
54 .output()
55 .expect("codewhale eval should start");
56
57 // Capture stdout/stderr for diagnostics
58 let stdout = String::from_utf8_lossy(&output.stdout);
59 let stderr = String::from_utf8_lossy(&output.stderr);
60
61 let report: Value = serde_json::from_str(&stdout).unwrap_or_else(|err| {
62 panic!("eval --json should emit valid JSON: {err}\nstdout:\n{stdout}\nstderr:\n{stderr}")
63 });
64
65 world.exit_status = Some(output.status);
66 world.report = Some(report);
67 }
68
69 #[then("the binary exits without crashing")]
70 fn binary_exits_without_crashing(world: &mut EvalSmokeWorld) {
71 let status = world
72 .exit_status
73 .expect("exit status should have been captured");
74
75 // The eval harness exits with code 1 when `metrics.success` is false
76 // (run_eval in main.rs uses `bail!("...")` for non-successful scenarios).
77 // This is expected behavior: the eval runs a multi-step scenario offline
78 // (List, Read, Search, Edit, ApplyPatch, ExecShell) and the overall
79 // metrics.success reflects all steps, not just Bash. The Bash
80 // step itself succeeds — see `json_report_contains_execution_steps`.
81 //
82 // What we verify here:
83 // 1. The process ran to completion (was killed by no signal)
84 // 2. A known exit code was produced (not a crash/hang)
85 // 3. Step-level success is validated by the next Gherkin step.
86 let exit_code = status.code().expect("process should have terminated");
87 assert_no_signal_crash(&status);
88 assert!(
89 exit_code == 0 || exit_code == 1,
90 "codewhale eval exited with unexpected code {exit_code} (expected 0 or 1)"
91 );
92
93 let report = world.report.as_ref().expect("eval report should exist");
94 let steps = report
95 .get("steps")
96 .and_then(|value| value.as_array())
97 .expect("eval report should have a 'steps' array");
98 assert!(
99 !steps.is_empty(),
100 "eval report should have at least one step"
101 );
102 }
103
104 #[then("the JSON report contains execution steps")]
105 fn json_report_contains_execution_steps(world: &mut EvalSmokeWorld) {
106 let report = world.report.as_ref().expect("eval report should exist");
107 let steps = report
108 .get("steps")
109 .and_then(|value| value.as_array())
110 .expect("eval report should have a 'steps' array");
111
112 // Find the Bash step and verify it contains the expected output
113 let exec_step = steps
114 .iter()
115 .find(|step| step.get("kind").and_then(|v| v.as_str()) == Some("Bash"))
116 .expect("eval report should have a Bash step");
117
118 let step_success = exec_step
119 .get("success")
120 .and_then(|v| v.as_bool())
121 .unwrap_or(false);
122 assert!(step_success, "Bash step should succeed, got: {exec_step:?}");
123
124 let output = exec_step
125 .get("output")
126 .and_then(|v| v.as_str())
127 .unwrap_or("");
128 assert!(
129 output.contains("eval-smoke-test"),
130 "Bash output should contain the shell command echo, got: {output}"
131 );
132 }
133
134 #[tokio::test(flavor = "current_thread")]
135 async fn eval_smoke_binary_loads_and_reports_steps() {
136 let writer = EvalSmokeWorld::cucumber()
137 .fail_on_skipped()
138 .with_default_cli()
139 .filter_run(FEATURE_PATH, move |feature, _, scenario| {
140 feature.name == FEATURE_NAME && scenario.name == SMOKE_SCENARIO
141 })
142 .await;
143 assert_eq!(
144 writer.failed_steps(),
145 0,
146 "scenario failed: {SMOKE_SCENARIO}"
147 );
148 assert_eq!(
149 writer.skipped_steps(),
150 0,
151 "scenario skipped steps: {SMOKE_SCENARIO}"
152 );
153 assert_eq!(
154 writer.passed_steps(),
155 4,
156 "scenario did not run: {SMOKE_SCENARIO}"
157 );
158 }
159
160 /// Assert the process was not killed by a signal (Unix-only check).
161 #[cfg(unix)]
162 fn assert_no_signal_crash(status: &ExitStatus) {
163 use std::os::unix::process::ExitStatusExt;
164 assert!(
165 status.signal().is_none(),
166 "codewhale eval was killed by signal {} (crash?)",
167 status.signal().unwrap()
168 );
169 }
170
171 /// No-op on non-Unix platforms where `ExitStatusExt` is unavailable.
172 #[cfg(not(unix))]
173 fn assert_no_signal_crash(_status: &ExitStatus) {}
174
174 lines RUST