返回 CodeWhale
debug_diagnostics_baseline_tests.rs
根目录 / crates / tui / src / commands / debug_diagnostics_baseline_tests.rs
1 //! FEAT-029 Phase 2: host-bound baseline parity for the `debug::diagnostics`
2 //! command slice.
3 //!
4 //! Every assertion in this module compares the *current* public command output
5 //! against a fixture captured from the untouched implementation at
6 //! `origin/main` `922679d6c0afe4556f3e5bc59073eb8e4cf93e07`. The fixtures are
7 //! hand-reviewed source; they are never regenerated from the migrated
8 //! implementation. Volatile report stamps, cache ages and owned temporary
9 //! paths are normalised only at the documented comparison boundary.
10 //!
11 //! The module deliberately lives at the `commands` root, outside
12 //! `groups/debug`, which FEAT-045 later moves into `codewhale-commands`.
13
14 use std::time::Instant;
15
16 use crate::commands::debug_diagnostics_test_support::{
17 DiagnosticsHarness, SealedHome, assert_fixture, normalize_cache_ages,
18 };
19 use crate::commands::{CommandResult, execute};
20 use crate::config::ProviderKind;
21 use crate::tui::app::{AppAction, TurnCacheRecord};
22 use codewhale_models::{ContentBlock, Message, Role, SystemPrompt};
23
24 fn render(result: &CommandResult) -> String {
25 let mut out = String::new();
26 out.push_str(&format!("is_error: {}\n", result.is_error));
27 match &result.message {
28 Some(message) => out.push_str(&format!("message:\n{message}\n")),
29 None => out.push_str("message: <none>\n"),
30 }
31 match &result.action {
32 Some(action) => out.push_str(&format!("action: {action:?}\n")),
33 None => out.push_str("action: <none>\n"),
34 }
35 out
36 }
37
38 /// Freeze the complete `/balance` branch pair: the supported provider emits the
39 /// `FetchBalance` action with no message, an unsupported provider emits the
40 /// exact message and no action. Asserting the whole `CommandResult` also pins
41 /// `is_error`, which is easy to regress independently of the visible text.
42 #[test]
43 fn balance_branch_pair_matches_baseline() {
44 let mut harness = DiagnosticsHarness::new();
45 harness.app.set_provider_identity_record(
46 crate::config::Config::default()
47 .resolve_provider_identity(ProviderKind::Deepseek.as_str())
48 .expect("captured fixture provider"),
49 );
50 assert_fixture(
51 "balance_supported.txt",
52 &render(&execute("/balance", &mut harness.app)),
53 );
54
55 let mut harness = DiagnosticsHarness::new();
56 harness.app.set_provider_identity_record(
57 crate::config::Config::default()
58 .resolve_provider_identity(ProviderKind::Ollama.as_str())
59 .expect("captured fixture provider"),
60 );
61 assert_fixture(
62 "balance_unsupported.txt",
63 &render(&execute("/balance", &mut harness.app)),
64 );
65 }
66
67 /// `/preview-request` is the one `Pure` command in the slice: its registered
68 /// handler must never build a host envelope. The frozen fixtures pin the parsed
69 /// action payloads and the exact diagnostics for every grammar branch.
70 #[test]
71 fn preview_request_grammar_matches_baseline() {
72 for (fixture_name, command) in [
73 ("preview_default.txt", "/preview-request"),
74 ("preview_json.txt", "/preview-request json"),
75 ("preview_manifest.txt", "/preview-request manifest"),
76 ("preview_base_prompt.txt", "/preview-request base-prompt"),
77 (
78 "preview_prompt.txt",
79 "/preview-request --prompt refactor the parser",
80 ),
81 (
82 "preview_prompt_json.txt",
83 "/preview-request json --prompt fix it",
84 ),
85 (
86 "preview_prompt_bytes.txt",
87 "/preview-request --prompt padded text ",
88 ),
89 ("preview_unknown.txt", "/preview-request nope"),
90 ("preview_bare_prompt.txt", "/preview-request --prompt"),
91 ("preview_conflict.txt", "/preview-request json base-prompt"),
92 ] {
93 let mut harness = DiagnosticsHarness::new();
94 assert_fixture(fixture_name, &render(&execute(command, &mut harness.app)));
95 }
96 }
97
98 /// The prompt is hashed into the previewed body, so its bytes must survive the
99 /// dispatcher verbatim — including leading interior and trailing whitespace —
100 /// and the command must not touch conversation state.
101 #[test]
102 fn preview_request_preserves_prompt_bytes_and_mutates_nothing() {
103 let mut harness = DiagnosticsHarness::new();
104 let before_messages = harness.app.api_messages.len();
105 let before_history = harness.app.history.len();
106
107 let prompt = " padded text ";
108 let result = execute(
109 &format!("/preview-request --prompt {prompt}"),
110 &mut harness.app,
111 );
112
113 let Some(AppAction::PreviewOutboundRequest {
114 json,
115 base_prompt_only,
116 hypothetical_prompt,
117 }) = result.action
118 else {
119 panic!("preview must emit the engine action: {result:?}");
120 };
121 assert!(!json);
122 assert!(!base_prompt_only);
123 // Exactly one whitespace codepoint delimits the flag; the rest are bytes.
124 assert_eq!(hypothetical_prompt.as_deref(), Some(prompt));
125 assert_eq!(harness.app.api_messages.len(), before_messages);
126 assert_eq!(harness.app.history.len(), before_history);
127 }
128
129 /// The snapshot-availability check happens *before* format validation: a
130 /// missing snapshot must report the truthful unavailable state for every
131 /// argument spelling, not the usage error.
132 #[test]
133 fn tools_checks_snapshot_before_format_validation() {
134 let mut harness = DiagnosticsHarness::new();
135 for (fixture_name, command) in [
136 ("tools_no_snapshot.txt", "/tools"),
137 ("tools_no_snapshot.txt", "/tools yaml"),
138 ] {
139 assert_fixture(fixture_name, &render(&execute(command, &mut harness.app)));
140 }
141 }
142
143 /// With a prepared snapshot present, the text, explicit-text, JSON and invalid
144 /// format branches are frozen byte-for-byte, and the compatibility alias
145 /// `/tool-studio` renders the same payload as the canonical name.
146 #[test]
147 fn tools_snapshot_branches_match_baseline() {
148 for (fixture_name, command) in [
149 ("tools_text.txt", "/tools"),
150 ("tools_text_explicit.txt", "/tools text"),
151 ("tools_json.txt", "/tools json"),
152 ("tools_invalid.txt", "/tools yaml"),
153 ] {
154 let mut harness = DiagnosticsHarness::new();
155 harness.app.session.last_tool_request_snapshot = Some(snapshot());
156 assert_fixture(fixture_name, &render(&execute(command, &mut harness.app)));
157 }
158
159 let mut harness = DiagnosticsHarness::new();
160 harness.app.session.last_tool_request_snapshot = Some(snapshot());
161 let canonical = execute("/tools json", &mut harness.app);
162 drop(harness);
163 let mut alias_harness = DiagnosticsHarness::new();
164 alias_harness.app.session.last_tool_request_snapshot = Some(snapshot());
165 let alias = execute("/tool-studio json", &mut alias_harness.app);
166 assert_eq!(
167 render(&canonical),
168 render(&alias),
169 "the /tool-studio alias must render the canonical payload"
170 );
171 }
172
173 /// `/system` Text/Blocks/None and the UTF-8-safe 500-byte truncation boundary
174 /// are frozen exactly, including the existing total-length wording.
175 #[test]
176 fn system_prompt_branches_match_baseline() {
177 let mut harness = DiagnosticsHarness::new();
178 harness.app.system_prompt = Some(codewhale_models::SystemPrompt::Text(
179 "You are a helpful assistant.".to_string(),
180 ));
181 assert_fixture(
182 "system_text.txt",
183 &render(&execute("/system", &mut harness.app)),
184 );
185
186 let mut harness = DiagnosticsHarness::new();
187 harness.app.system_prompt = Some(codewhale_models::SystemPrompt::Blocks(vec![
188 codewhale_models::SystemBlock {
189 block_type: "text".to_string(),
190 text: "First block".to_string(),
191 cache_control: None,
192 },
193 codewhale_models::SystemBlock {
194 block_type: "text".to_string(),
195 text: "Second block".to_string(),
196 cache_control: None,
197 },
198 ]));
199 assert_fixture(
200 "system_blocks.txt",
201 &render(&execute("/system", &mut harness.app)),
202 );
203
204 let mut harness = DiagnosticsHarness::new();
205 harness.app.system_prompt = None;
206 assert_fixture(
207 "system_none.txt",
208 &render(&execute("/system", &mut harness.app)),
209 );
210
211 let mut harness = DiagnosticsHarness::new();
212 harness.app.system_prompt = Some(codewhale_models::SystemPrompt::Text(format!(
213 "{}\u{e9}\u{4e2d}",
214 "x".repeat(520)
215 )));
216 assert_fixture(
217 "system_truncated.txt",
218 &render(&execute("/system", &mut harness.app)),
219 );
220 }
221
222 /// The `/system` alias `/xitong` is the same registration; the truncation
223 /// boundary must stay UTF-8 safe at the existing 500-byte cut.
224 #[test]
225 fn system_alias_and_truncation_boundary_are_preserved() {
226 let mut harness = DiagnosticsHarness::new();
227 harness.app.system_prompt = Some(codewhale_models::SystemPrompt::Text(format!(
228 "{}\u{e9}\u{4e2d}",
229 "x".repeat(520)
230 )));
231 let canonical = execute("/system", &mut harness.app);
232 let alias = execute("/xitong", &mut harness.app);
233 assert_eq!(render(&canonical), render(&alias));
234 let message = canonical.message.expect("system message");
235 assert!(
236 message.contains("(truncated, 525 chars total)"),
237 "{message}"
238 );
239 // The cut lands before the two multibyte characters, so no replacement
240 // character from an invalid boundary may appear.
241 assert!(!message.contains('\u{fffd}'), "{message}");
242 }
243
244 /// `/context` bare opens the inspector; the report subcommands delegate to the
245 /// shared renderer; an unknown subcommand is an exact error with no action.
246 #[test]
247 fn context_routing_and_report_branches_match_baseline() {
248 // The report counts the user's global instructions and installed skills.
249 let _home = SealedHome::new();
250 let mut harness = DiagnosticsHarness::new();
251 assert_fixture(
252 "context_bare.txt",
253 &render(&execute("/context", &mut harness.app)),
254 );
255
256 let mut harness = DiagnosticsHarness::new();
257 assert_fixture(
258 "context_unknown.txt",
259 &render(&execute("/context bogus", &mut harness.app)),
260 );
261
262 for (fixture_name, command) in [
263 ("context_report.txt", "/context report"),
264 ("context_json.txt", "/context json"),
265 ("context_summary.txt", "/context summary"),
266 ("context_prompt_json.txt", "/context prompt-json"),
267 ] {
268 let mut harness = DiagnosticsHarness::new();
269 let rendered = render(&execute(command, &mut harness.app));
270 assert_fixture(fixture_name, &harness.normalize(&rendered));
271 }
272 }
273
274 /// The `/context` alias `/ctx` resolves to the same entry, and the bare form
275 /// must return the inspector action with no message.
276 #[test]
277 fn context_alias_and_bare_action_are_preserved() {
278 // The alias and canonical reports are compared byte for byte, and both
279 // enumerate the user's installed skills: on a developer machine their
280 // discovery can differ between two calls in one test.
281 let _home = SealedHome::new();
282 let mut harness = DiagnosticsHarness::new();
283 let bare = execute("/context", &mut harness.app);
284 assert!(bare.message.is_none());
285 assert!(matches!(bare.action, Some(AppAction::OpenContextInspector)));
286
287 let alias = execute("/ctx report", &mut harness.app);
288 let canonical = execute("/context report", &mut harness.app);
289 assert_eq!(
290 alias.message.as_deref().unwrap_or_default(),
291 canonical.message.as_deref().unwrap_or_default(),
292 );
293 }
294
295 /// `/tokens` and `/cost` on both the empty and seeded sessions are frozen
296 /// exactly, including the estimate disclaimer and the coverage line that the
297 /// two surfaces share.
298 #[test]
299 fn tokens_and_cost_coverage_surfaces_match_baseline() {
300 let mut harness = DiagnosticsHarness::new();
301 assert_fixture(
302 "tokens_empty.txt",
303 &render(&execute("/tokens", &mut harness.app)),
304 );
305 assert_fixture(
306 "cost_empty.txt",
307 &render(&execute("/cost", &mut harness.app)),
308 );
309
310 let mut harness = DiagnosticsHarness::new();
311 harness.app.session.total_tokens = 1234;
312 harness.app.session.session_cost = 0.05;
313 harness.app.session.cost_priced_turns = 1;
314 harness.app.session.last_prompt_tokens = Some(100);
315 harness.app.session.last_completion_tokens = Some(25);
316 harness.app.session.last_prompt_cache_hit_tokens = Some(70);
317 harness.app.session.last_prompt_cache_miss_tokens = Some(30);
318 assert_fixture(
319 "tokens_seeded.txt",
320 &render(&execute("/tokens", &mut harness.app)),
321 );
322 assert_fixture(
323 "cost_seeded.txt",
324 &render(&execute("/cost", &mut harness.app)),
325 );
326 }
327
328 /// `/cache` without telemetry, the unknown-argument error, the warmup action,
329 /// the empty stats/zones reports and the inspect conflict branch are frozen.
330 #[test]
331 fn cache_static_branches_match_baseline() {
332 for (fixture_name, command) in [
333 ("cache_no_data.txt", "/cache"),
334 ("cache_unknown.txt", "/cache wat"),
335 ("cache_warmup.txt", "/cache warmup"),
336 ("cache_stats_empty.txt", "/cache stats"),
337 ("cache_zones_empty.txt", "/cache zones"),
338 ("cache_inspect_empty.txt", "/cache inspect"),
339 (
340 "cache_inspect_conflict.txt",
341 "/cache inspect --json --verbose",
342 ),
343 ] {
344 let mut harness = DiagnosticsHarness::new();
345 assert_fixture(fixture_name, &render(&execute(command, &mut harness.app)));
346 }
347
348 // The conflict branch is a bounded message, not a hard error.
349 let mut harness = DiagnosticsHarness::new();
350 let conflict = execute("/cache inspect --json --verbose", &mut harness.app);
351 assert!(!conflict.is_error);
352 assert_eq!(
353 conflict.message.as_deref(),
354 Some("cache inspect: --json and --verbose cannot be combined")
355 );
356 }
357
358 /// Recorded turn telemetry is frozen for the history table (full and
359 /// count-clamped) and for the stats/zones aggregations; only the per-turn age
360 /// cell is normalised.
361 #[test]
362 fn cache_recorded_telemetry_matches_baseline() {
363 for (fixture_name, command) in [
364 ("cache_history.txt", "/cache"),
365 ("cache_history_count.txt", "/cache 2"),
366 ("cache_stats_seeded.txt", "/cache stats"),
367 ("cache_zones_seeded.txt", "/cache zones"),
368 ] {
369 let mut harness = DiagnosticsHarness::new();
370 seed_turns(&mut harness);
371 let rendered = render(&execute(command, &mut harness.app));
372 assert_fixture(fixture_name, &normalize_cache_ages(&rendered));
373 }
374 }
375
376 /// Freeze the ordered JSON bytes and the observation commit after a real
377 /// successful inspection. A second inspection compares with the stored first
378 /// request after the system prefix changes.
379 #[test]
380 fn cache_inspect_json_and_previous_state_match_baseline() {
381 let mut harness = DiagnosticsHarness::new();
382 harness.app.system_prompt = Some(SystemPrompt::Text("Base policy".into()));
383 harness.app.session.last_tool_catalog = Some(vec![tool("read_file")]);
384 harness.app.api_messages_mut().push(Message {
385 role: Role::User,
386 content: vec![ContentBlock::Text {
387 text: "Current task".into(),
388 cache_control: None,
389 }],
390 });
391 assert!(harness.app.session.last_cache_inspection.is_none());
392 let first = execute("/cache inspect --json", &mut harness.app);
393 assert!(harness.app.session.last_cache_inspection.is_some());
394 let previous = harness.app.session.last_cache_inspection.clone();
395 assert_fixture("cache_inspect_json.txt", &render(&first));
396
397 harness.app.system_prompt = Some(SystemPrompt::Text("Changed policy".into()));
398 let second = execute("/cache inspect --json", &mut harness.app);
399 assert_ne!(harness.app.session.last_cache_inspection, previous);
400 assert_fixture("cache_inspect_json_changed.txt", &render(&second));
401 }
402
403 /// The slice inventory is exactly the eight declared commands plus the five
404 /// mutation commands now adopted by the same debug group; no diagnostics command may
405 /// lose its registration.
406 #[test]
407 fn diagnostics_slice_inventory_is_exactly_the_declared_eight() {
408 for name in [
409 "tokens",
410 "cost",
411 "balance",
412 "cache",
413 "preview-request",
414 "tools",
415 "system",
416 "context",
417 ] {
418 assert!(
419 crate::commands::get_command_info(name).is_some(),
420 "/{name} must remain registered"
421 );
422 }
423 for name in ["change", "edit", "diff", "undo", "retry"] {
424 assert!(
425 crate::commands::get_command_info(name).is_some(),
426 "/{name} is the FEAT-030 mutation slice and must stay registered"
427 );
428 }
429 }
430
431 fn snapshot() -> crate::tool_inspection::ToolInspectionSnapshot {
432 crate::tool_inspection::ToolInspectionSnapshot::from_prepared_request(
433 "turn-1",
434 2,
435 Some(&[tool("read_file"), tool("write_file")]),
436 )
437 }
438
439 fn seed_turns(harness: &mut DiagnosticsHarness) {
440 let now = Instant::now();
441 for record in [
442 turn_record(4_000, 200, Some(3_000), Some(1_000), None, now),
443 turn_record(6_000, 250, Some(3_000), Some(3_000), Some(150), now),
444 turn_record(5_000, 100, Some(2_500), None, None, now),
445 turn_record(1_000, 50, None, None, None, now),
446 ] {
447 harness.app.push_turn_cache_record(record);
448 }
449 }
450
451 fn turn_record(
452 input_tokens: u32,
453 output_tokens: u32,
454 cache_hit_tokens: Option<u32>,
455 cache_miss_tokens: Option<u32>,
456 reasoning_replay_tokens: Option<u32>,
457 recorded_at: Instant,
458 ) -> TurnCacheRecord {
459 TurnCacheRecord {
460 provider: Some(ProviderKind::Deepseek),
461 provider_identity: Some("deepseek".to_string()),
462 model: Some("deepseek-v4-pro".to_string()),
463 auto_model: false,
464 input_tokens,
465 output_tokens,
466 cache_hit_tokens,
467 cache_miss_tokens,
468 reasoning_replay_tokens,
469 cache_write_tokens: None,
470 reasoning_tokens: None,
471 cost_audit: None,
472 recorded_at,
473 }
474 }
475
476 fn tool(name: &str) -> codewhale_models::Tool {
477 codewhale_models::Tool {
478 tool_type: Some("function".to_string()),
479 name: name.to_string(),
480 description: format!("{name} test tool"),
481 input_schema: serde_json::json!({
482 "type": "object",
483 "properties": {"path": {"type": "string"}}
484 }),
485 allowed_callers: None,
486 defer_loading: Some(false),
487 input_examples: None,
488 strict: Some(true),
489 cache_control: None,
490 }
491 }
492
492 lines RUST