| 1 | //! Gherkin binary health and eval harness smoke test for command extraction. |
| 2 | //! |
| 3 | //! This runs the binary through `codewhale eval` and verifies that the |
| 4 | //! executable still loads and reports a successful JSON evaluation after the |
| 5 | //! core/session command modules are extracted. |
| 6 | |
| 7 | use std::process::Command; |
| 8 | |
| 9 | use cucumber::{World as _, given, then, when, writer::Stats as _}; |
| 10 | use serde_json::Value; |
| 11 | use tempfile::TempDir; |
| 12 | |
| 13 | const FEATURE_NAME: &str = "Core and session command extraction"; |
| 14 | const FEATURE_PATH: &str = concat!( |
| 15 | env!("CARGO_MANIFEST_DIR"), |
| 16 | "/tests/features/core_session_command_extraction.feature" |
| 17 | ); |
| 18 | const CORE_SCENARIO: &str = "The binary loads and runs the evaluation harness after extraction"; |
| 19 | |
| 20 | #[derive(Debug, Default, cucumber::World)] |
| 21 | struct CoreSessionExtractionWorld { |
| 22 | record_dir: Option<TempDir>, |
| 23 | report: Option<Value>, |
| 24 | } |
| 25 | |
| 26 | #[given("a clean CodeWhale evaluation workspace")] |
| 27 | fn clean_codewhale_evaluation_workspace(world: &mut CoreSessionExtractionWorld) { |
| 28 | world.record_dir = Some(TempDir::new().expect("evaluation TempDir")); |
| 29 | } |
| 30 | |
| 31 | #[when("the evaluation harness runs a shell command")] |
| 32 | fn eval_harness_runs_shell_command(world: &mut CoreSessionExtractionWorld) { |
| 33 | let record_dir = world |
| 34 | .record_dir |
| 35 | .as_ref() |
| 36 | .expect("evaluation workspace should exist"); |
| 37 | |
| 38 | let output = Command::new(crate::binary::codewhale()) |
| 39 | .args([ |
| 40 | "eval", |
| 41 | "--json", |
| 42 | "--shell-command", |
| 43 | "echo eval-harness", |
| 44 | "--record", |
| 45 | ]) |
| 46 | .arg(record_dir.path()) |
| 47 | .output() |
| 48 | .expect("codewhale eval should start"); |
| 49 | |
| 50 | assert!( |
| 51 | output.status.success(), |
| 52 | "codewhale eval failed\nstderr:\n{}", |
| 53 | String::from_utf8_lossy(&output.stderr) |
| 54 | ); |
| 55 | |
| 56 | let report: Value = serde_json::from_slice(&output.stdout).unwrap_or_else(|err| { |
| 57 | panic!( |
| 58 | "eval --json should emit valid JSON: {err}\nstdout:\n{}", |
| 59 | String::from_utf8_lossy(&output.stdout) |
| 60 | ) |
| 61 | }); |
| 62 | |
| 63 | world.report = Some(report); |
| 64 | } |
| 65 | |
| 66 | #[then("the harness completes successfully")] |
| 67 | fn harness_completes_successfully(world: &mut CoreSessionExtractionWorld) { |
| 68 | let report = world.report.as_ref().expect("eval report should exist"); |
| 69 | |
| 70 | let success = report |
| 71 | .get("metrics") |
| 72 | .and_then(|metrics| metrics.get("success")) |
| 73 | .and_then(|value| value.as_bool()) |
| 74 | .unwrap_or(false); |
| 75 | assert!( |
| 76 | success, |
| 77 | "eval report 'metrics.success' should be true, got: {report:?}" |
| 78 | ); |
| 79 | } |
| 80 | |
| 81 | #[then("the JSON report contains a step with the expected kind")] |
| 82 | fn json_report_contains_step_with_expected_kind(world: &mut CoreSessionExtractionWorld) { |
| 83 | let report = world.report.as_ref().expect("eval report should exist"); |
| 84 | |
| 85 | let steps = report |
| 86 | .get("steps") |
| 87 | .and_then(|value| value.as_array()) |
| 88 | .expect("eval report should have a 'steps' array"); |
| 89 | |
| 90 | assert!( |
| 91 | !steps.is_empty(), |
| 92 | "eval report should have at least one step" |
| 93 | ); |
| 94 | |
| 95 | let first_step = &steps[0]; |
| 96 | let kind = first_step |
| 97 | .get("kind") |
| 98 | .and_then(|value| value.as_str()) |
| 99 | .expect("step should have a 'kind' field"); |
| 100 | |
| 101 | assert_eq!( |
| 102 | kind, "List", |
| 103 | "first step kind should be 'List', got: {kind}" |
| 104 | ); |
| 105 | |
| 106 | let step_success = first_step |
| 107 | .get("success") |
| 108 | .and_then(|value| value.as_bool()) |
| 109 | .unwrap_or(false); |
| 110 | assert!( |
| 111 | step_success, |
| 112 | "first step 'success' should be true, got: {first_step:?}" |
| 113 | ); |
| 114 | |
| 115 | let output = first_step |
| 116 | .get("output") |
| 117 | .and_then(|value| value.as_str()) |
| 118 | .unwrap_or(""); |
| 119 | assert!( |
| 120 | !output.is_empty(), |
| 121 | "step output should not be empty: {first_step:?}" |
| 122 | ); |
| 123 | } |
| 124 | |
| 125 | #[tokio::test(flavor = "current_thread")] |
| 126 | async fn codewhale_eval_runs_after_extraction() { |
| 127 | let writer = CoreSessionExtractionWorld::cucumber() |
| 128 | .fail_on_skipped() |
| 129 | .with_default_cli() |
| 130 | .filter_run(FEATURE_PATH, move |feature, _, scenario| { |
| 131 | feature.name == FEATURE_NAME && scenario.name == CORE_SCENARIO |
| 132 | }) |
| 133 | .await; |
| 134 | assert_eq!(writer.failed_steps(), 0, "scenario failed: {CORE_SCENARIO}"); |
| 135 | assert_eq!( |
| 136 | writer.skipped_steps(), |
| 137 | 0, |
| 138 | "scenario skipped steps: {CORE_SCENARIO}" |
| 139 | ); |
| 140 | assert_eq!( |
| 141 | writer.passed_steps(), |
| 142 | 4, |
| 143 | "scenario did not run: {CORE_SCENARIO}" |
| 144 | ); |
| 145 | } |
| 146 |