返回 CodeWhale
tests.rs
根目录 / crates / tui / src / tui / history / tests.rs
1 //! Transcript history-cell tests.
2 //!
3 //! Rebuilt in v0.9.11 after declaring test bankruptcy on the previous suite
4 //! (123 tests / 3,964 lines). About a third of that file pinned glyph choices,
5 //! palette tokens, span indices and English label text — `spans[1] == "⣤"`,
6 //! `title_span.style.fg == theme.tool_title_color`, `visible[1] == "▏ done:
7 //! scan repo"`. Those assertions fail on every legitimate visual refactor and
8 //! catch nothing a user would notice, which is the liability `d64b9429b`
9 //! ("remove brittle visual test mass") named.
10 //!
11 //! What survives is named for the *property* it protects. Rules for additions:
12 //!
13 //! * Assert a property, not a token. `spans[1] == "⣤"` is a token; "the frame
14 //! does not change while motion is reduced" is the property, and it is
15 //! strictly stronger — it also catches an animation leak the constant missed.
16 //! * Where the value is a design choice (color, glyph, verb), assert the
17 //! *relationship* between cases instead: warning must not read as error.
18 //! * One test per property, with its cases in a table — not one test per case.
19 //! * Never assert `a || b` where `b` is trivially true of any English string.
20
21 use super::constants::{TOOL_OUTPUT_HEAD_LINES, TOOL_OUTPUT_LINE_LIMIT, TOOL_OUTPUT_TAIL_LINES};
22 use super::thinking::cached_color_depth;
23 use super::{
24 ASSISTANT_GLYPH, ExecCell, ExecSource, GenericToolCell, HistoryCell, PlanUpdateCell,
25 REASONING_CURSOR, REASONING_OPENER, REASONING_RAIL, RenderMode, ThinkingFold, ToolCell,
26 ToolStatus, TranscriptRenderOptions, WebSearchCell, assistant_label_style_for,
27 extract_reasoning_summary, render_spillover_annotation, render_thinking,
28 render_thinking_with_analysis, running_status_label_with_elapsed,
29 };
30 use crate::tools::plan::{PlanSnapshot, StepStatus};
31 use crate::tui::motion::MotionMode;
32 use crate::tui::ui_text::{
33 line_to_plain, slice_visible_columns, text_display_width, text_visible_width,
34 };
35 use codewhale_models::{ContentBlock, Message, Role};
36 use std::path::PathBuf;
37 use std::time::{Duration, Instant};
38
39 #[test]
40 fn extension_prompt_snapshots_and_withdrawal_show_the_complete_model_input() {
41 let instructions = format!("{}\nLast instruction.", "x".repeat(12_000));
42 for block in [Some(instructions.as_str()), None] {
43 let message = crate::runtime_handoff::extension_prompt_contributions_runtime_message(block);
44 let cells = super::history_cells_from_message(&message);
45 let [HistoryCell::System { content }] = cells.as_slice() else {
46 panic!("runtime instructions must be shown as a system receipt");
47 };
48 match block {
49 Some(text) => assert!(
50 content.contains(text),
51 "the full model-visible text stays auditable"
52 ),
53 None => assert!(content.contains("withdrawn")),
54 }
55 }
56 }
57
58 // ---------------------------------------------------------------------------
59 // Helpers
60 // ---------------------------------------------------------------------------
61
62 fn line_text(line: &ratatui::text::Line<'static>) -> String {
63 line.spans
64 .iter()
65 .map(|span| span.content.as_ref())
66 .collect()
67 }
68
69 fn lines_text(lines: &[ratatui::text::Line<'static>]) -> String {
70 lines.iter().map(line_text).collect::<Vec<_>>().join("\n")
71 }
72
73 fn generic_tool(name: &str, status: ToolStatus) -> GenericToolCell {
74 GenericToolCell {
75 name: name.to_string(),
76 status,
77 input_summary: None,
78 output: None,
79 prompts: None,
80 spillover_path: None,
81 output_summary: None,
82 is_diff: false,
83 }
84 }
85
86 fn exec_tool(command: &str, status: ToolStatus) -> ExecCell {
87 ExecCell {
88 command: command.to_string(),
89 status,
90 output: None,
91 live_output: None,
92 shell_task_id: None,
93 owner_agent_id: None,
94 owner_agent_name: None,
95 started_at: None,
96 duration_ms: None,
97 stale_elapsed_since_output_ms: None,
98 source: ExecSource::Assistant,
99 interaction: None,
100 output_summary: None,
101 }
102 }
103
104 fn numbered_output(count: usize) -> String {
105 (0..count)
106 .map(|i| format!("row {i:02} plain content"))
107 .collect::<Vec<_>>()
108 .join("\n")
109 }
110
111 fn calm_options() -> TranscriptRenderOptions {
112 TranscriptRenderOptions {
113 low_motion: true,
114 ..TranscriptRenderOptions::default()
115 }
116 }
117
118 // ---------------------------------------------------------------------------
119 // Leaks — a rendered cell never exposes something the user was not shown
120 // ---------------------------------------------------------------------------
121
122 /// Spilled tool output lives in a file under the session directory. The path is
123 /// an internal storage detail: it names the user's home, their session id, and
124 /// a content hash, and it is useless to them because the affordance opens the
125 /// pager, not the file. No width, no render mode, and no standalone annotation
126 /// may print it.
127 ///
128 /// Replaces five separate tests that each checked one width or one mode.
129 #[test]
130 fn no_width_or_render_mode_leaks_a_spillover_storage_path() {
131 let secret = "/Users/private/.codewhale/sessions/session-a/artifacts/hash.txt";
132
133 for width in [18_u16, 40, 80, 120] {
134 for mode in [RenderMode::Live, RenderMode::Transcript] {
135 let mut cell = generic_tool("read_file", ToolStatus::Success);
136 cell.input_summary = Some("cmd: cargo build --release".to_string());
137 cell.output = Some(numbered_output(40));
138 cell.spillover_path = Some(PathBuf::from(secret));
139
140 let rendered = lines_text(&cell.lines_with_mode(width, true, mode));
141 for fragment in ["/Users", ".codewhale", "sessions/", "hash.txt"] {
142 assert!(
143 !rendered.contains(fragment),
144 "storage path fragment {fragment:?} leaked at width {width} in {mode:?}: \
145 {rendered:?}"
146 );
147 }
148 }
149
150 // The standalone affordance carries no path either, and it fits the
151 // width it was given — an affordance that overflows is a wrap artifact
152 // in the transcript.
153 let annotation = line_to_plain(&render_spillover_annotation(width));
154 assert!(
155 text_display_width(&annotation) <= usize::from(width),
156 "affordance exceeds width {width}: {annotation:?}"
157 );
158 for fragment in ["/Users", ".codewhale", "hash.txt"] {
159 assert!(
160 !annotation.contains(fragment),
161 "affordance leaked {fragment:?}: {annotation:?}"
162 );
163 }
164 }
165
166 // The common case — a result that never spilled — spends no row on an
167 // affordance that would open an empty pager.
168 let mut plain = generic_tool("read_file", ToolStatus::Success);
169 plain.output = Some("contents".to_string());
170 let hint = crate::tui::key_shortcuts::tool_details_shortcut_action_hint("output");
171 let rendered = lines_text(&plain.lines_with_mode(80, true, RenderMode::Live));
172 assert!(
173 !rendered.contains(&hint),
174 "a result that did not spill must not advertise the spillover pager: {rendered:?}"
175 );
176 }
177
178 /// With reasoning display off, the model's chain of thought must not reach the
179 /// screen in any lifecycle state — not while streaming, not once complete.
180 /// The live case still needs a progress signal, so it renders one compact row.
181 #[test]
182 fn hidden_reasoning_never_renders_its_content_in_any_state() {
183 let secret = "private chain of thought that must not be shown";
184 let hidden = TranscriptRenderOptions {
185 show_thinking: false,
186 low_motion: true,
187 ..TranscriptRenderOptions::default()
188 };
189
190 let streaming = HistoryCell::Thinking {
191 content: secret.to_string(),
192 streaming: true,
193 duration_secs: None,
194 };
195 let live = streaming.lines_with_options(80, hidden);
196 let live_text = lines_text(&live);
197 assert!(
198 !live_text.contains(secret),
199 "hidden live reasoning revealed its body: {live_text}"
200 );
201 assert_eq!(
202 live.len(),
203 1,
204 "hidden reasoning is one compact progress row, not a stack of state \
205 copy: {live_text}"
206 );
207
208 let complete = HistoryCell::Thinking {
209 content: secret.to_string(),
210 streaming: false,
211 duration_secs: Some(1.0),
212 };
213 assert!(
214 complete.lines_with_options(80, hidden).is_empty(),
215 "completed hidden reasoning must leave the live transcript entirely"
216 );
217 }
218
219 /// A live card is a summary. It must name the tool that ran (so the row is
220 /// attributable) and must not spend its one line echoing arguments the caller
221 /// never chose — `max_count: 15` is a schema default, not a user intent.
222 /// Transcript replay is the record, so it keeps the exact tool id.
223 ///
224 /// Replaces six tests, one of which (`unknown_generic_tool_keeps_raw_name_in_
225 /// live_mode`) asserted only `!text.is_empty()` and so could never fail.
226 #[test]
227 fn live_cards_name_their_tool_without_echoing_control_only_arguments() {
228 assert_eq!(
229 super::summarize_tool_args(&serde_json::json!({
230 "max_count": 15,
231 "timeout_ms": 30_000
232 })),
233 None,
234 "an argument set that is entirely control defaults summarizes to nothing"
235 );
236 assert_eq!(
237 super::summarize_tool_args(&serde_json::json!({
238 "max_count": 15,
239 "branch": "main"
240 }))
241 .as_deref(),
242 Some("branch: main"),
243 "the meaningful key is what the summary is for"
244 );
245
246 for name in ["git_log", "future_private_tool"] {
247 let mut cell = generic_tool(name, ToolStatus::Success);
248 cell.input_summary = Some("max_count: 15".to_string());
249 let lines = cell.lines_with_mode(120, true, RenderMode::Live);
250 let joined = lines_text(&lines);
251
252 assert_eq!(lines.len(), 1, "compact live row for {name}: {joined:?}");
253 assert!(
254 joined.contains(name),
255 "the row must be attributable to {name}: {joined:?}"
256 );
257 assert!(
258 !joined.contains("max_count"),
259 "control defaults must not become the visible summary for {name}: {joined:?}"
260 );
261 }
262
263 // A tool the UI has a family for is identified by that family live; the
264 // raw id would be a second, redundant name. Replay keeps it.
265 let mut known = generic_tool("run_verifiers", ToolStatus::Running);
266 known.input_summary = Some("profile: auto, level: quick".to_string());
267 let known = HistoryCell::Tool(ToolCell::Generic(known));
268 let live = lines_text(&known.lines(80));
269 let transcript = lines_text(&known.transcript_lines(80));
270 assert!(
271 !live.contains("run_verifiers"),
272 "a known tool id must not take a slot in the compact live card: {live}"
273 );
274 assert!(
275 transcript.contains("run_verifiers"),
276 "transcript replay preserves the exact tool id: {transcript}"
277 );
278 }
279
280 // ---------------------------------------------------------------------------
281 // Budget — the card never lies about how much it is showing
282 // ---------------------------------------------------------------------------
283
284 /// `selected_output_indices` fills head + tail and then tops up from lines that
285 /// look important (error / warning / path). Plain output — a list of names, a
286 /// clean build log — matches none of those, so the top-up found nothing and the
287 /// card silently forfeited the rest of its budget while still reporting the
288 /// remainder as omitted. A card that advertises N rows shows N rows.
289 #[test]
290 fn a_live_card_spends_the_whole_output_budget_it_advertises() {
291 let total = 40usize;
292 let cell = {
293 let mut exec = exec_tool("list_things", ToolStatus::Failed);
294 exec.output = Some(numbered_output(total));
295 exec.duration_ms = Some(120);
296 HistoryCell::Tool(ToolCell::Exec(exec))
297 };
298
299 let live_text = lines_text(&cell.lines_with_options(80, calm_options()));
300 let shown = (0..total)
301 .filter(|i| live_text.contains(&format!("row {i:02} plain content")))
302 .count();
303
304 assert_eq!(
305 shown, TOOL_OUTPUT_LINE_LIMIT,
306 "a card promising {TOOL_OUTPUT_LINE_LIMIT} rows must show \
307 {TOOL_OUTPUT_LINE_LIMIT}, not stop at head+tail: {live_text}"
308 );
309 for i in 0..TOOL_OUTPUT_HEAD_LINES {
310 assert!(
311 live_text.contains(&format!("row {i:02} plain content")),
312 "head row {i} missing: {live_text}"
313 );
314 }
315 for i in (total - TOOL_OUTPUT_TAIL_LINES)..total {
316 assert!(
317 live_text.contains(&format!("row {i:02} plain content")),
318 "tail row {i} missing: {live_text}"
319 );
320 }
321 }
322
323 /// Failure keeps the invocation plus head/tail evidence visible under every
324 /// density setting; the full record remains in the details view.
325 #[test]
326 fn calm1_failed_tool_keeps_context_and_result_under_every_density_setting() {
327 let total = 30usize;
328 let last = format!("row {:02} plain content", total - 1);
329
330 for (label, options) in [
331 ("default", TranscriptRenderOptions::default()),
332 (
333 "tool details hidden",
334 TranscriptRenderOptions {
335 show_tool_details: false,
336 ..TranscriptRenderOptions::default()
337 },
338 ),
339 (
340 "calm mode",
341 TranscriptRenderOptions {
342 calm_mode: true,
343 ..TranscriptRenderOptions::default()
344 },
345 ),
346 ] {
347 let cell = {
348 let mut cell = generic_tool("read_file", ToolStatus::Failed);
349 cell.input_summary = Some("command: noisy".to_string());
350 cell.output = Some(numbered_output(total));
351 HistoryCell::Tool(ToolCell::Generic(cell))
352 };
353
354 let text = lines_text(&cell.lines_with_options(80, options));
355 assert!(
356 text.contains("lines omitted"),
357 "[{label}] a bounded failure must advertise omitted output: {text}"
358 );
359 assert!(
360 text.contains(&last),
361 "[{label}] failed output must stay expanded to its last row: {text}"
362 );
363 assert!(
364 text.contains("command: noisy"),
365 "[{label}] the failing invocation must stay visible: {text}"
366 );
367 }
368 }
369
370 /// The live surface is a summary and the transcript is the record. The contract
371 /// is directional: anything the live view drops must still be in the
372 /// transcript, and the live view must say so when it drops something. A success
373 /// gets a bounded preview; a failure gets the full budget; neither may leave
374 /// the transcript short.
375 ///
376 /// Replaces four near-identical live/transcript comparison tests.
377 #[test]
378 fn whatever_live_truncates_the_transcript_still_holds() {
379 let total = 30usize;
380 let first = "row 00 plain content";
381 let last = format!("row {:02} plain content", total - 1);
382
383 // Failed exec: capped live with an honest marker, uncapped in transcript.
384 let failed = {
385 let mut exec = exec_tool("noisy_script.sh", ToolStatus::Failed);
386 exec.output = Some(numbered_output(total));
387 exec.duration_ms = Some(120);
388 HistoryCell::Tool(ToolCell::Exec(exec))
389 };
390 let live = failed.lines_with_options(80, calm_options());
391 let transcript = failed.transcript_lines(80);
392 let live_text = lines_text(&live);
393 let transcript_text = lines_text(&transcript);
394 assert!(
395 live.len() < transcript.len(),
396 "live must compress (live={}, transcript={})",
397 live.len(),
398 transcript.len()
399 );
400 assert!(
401 live_text.contains("lines omitted"),
402 "a live view that drops rows must say so: {live_text}"
403 );
404 assert!(
405 !transcript_text.contains("lines omitted"),
406 "the transcript drops nothing, so it claims nothing: {transcript_text}"
407 );
408 assert!(transcript_text.contains(first) && transcript_text.contains(&last));
409 assert!(
410 transcript_text.contains("row 15 plain content"),
411 "the transcript keeps the middle the live view skipped: {transcript_text}"
412 );
413
414 // Successful exec: a bounded head preview, never the whole body.
415 let success = {
416 let mut exec = exec_tool("noisy_script.sh", ToolStatus::Success);
417 exec.output = Some(numbered_output(total));
418 exec.duration_ms = Some(120);
419 HistoryCell::Tool(ToolCell::Exec(exec))
420 };
421 let live_text = lines_text(&success.lines_with_options(80, calm_options()));
422 let transcript_text = lines_text(&success.transcript_lines(80));
423 let previewed = (0..total)
424 .filter(|i| live_text.contains(&format!("row {i:02} plain content")))
425 .count();
426 assert_eq!(
427 previewed, 3,
428 "a successful exec previews only the opening and final result rows: {live_text}"
429 );
430 assert!(
431 live_text.contains(first) && live_text.contains(&last),
432 "the bounded preview retains context and the final result: {live_text}"
433 );
434 let output_hint = crate::tui::key_shortcuts::tool_details_shortcut_action_hint("output");
435 assert!(
436 live_text.contains(&output_hint),
437 "a shortened success must point to its full output: {live_text}"
438 );
439 assert!(
440 !live_text.contains("row 15 plain content"),
441 "the live preview should leave routine middle output for details: {live_text}"
442 );
443 assert!(transcript_text.contains(first) && transcript_text.contains(&last));
444 assert!(
445 transcript_text.contains("row 15 plain content"),
446 "the full transcript must retain output omitted from the live card: {transcript_text}"
447 );
448
449 // Successful generic tool: output collapses entirely live, and does so
450 // without spending a row telling the user it collapsed.
451 let quiet = {
452 let mut cell = generic_tool("read_file", ToolStatus::Success);
453 cell.input_summary = Some("path: crates/tui/src/main.rs".to_string());
454 cell.output = Some(numbered_output(24));
455 HistoryCell::Tool(ToolCell::Generic(cell))
456 };
457 let live_text = lines_text(&quiet.lines_with_options(80, TranscriptRenderOptions::default()));
458 let transcript_text = lines_text(&quiet.transcript_lines(80));
459 assert!(
460 !live_text.contains(first) && !live_text.contains("lines omitted"),
461 "a quiet success collapses silently: {live_text}"
462 );
463 assert!(transcript_text.contains(first));
464 assert!(transcript_text.contains("row 23 plain content"));
465 }
466
467 /// Repro for #80: a `git diff --stat`-shaped result must keep its newlines on
468 /// the transcript surface — one file per row, not squashed into one line.
469 #[test]
470 fn multi_line_tool_output_keeps_one_row_per_source_line() {
471 let diff_stat = "Cargo.lock | 1 +\n\
472 crates/cli/Cargo.toml | 1 +\n\
473 crates/cli/src/main.rs | 47 ++++++\n\
474 crates/config/src/lib.rs | 27 ++++\n\
475 crates/tui/src/mcp.rs | 384 +++++";
476
477 let cell = {
478 let mut cell = generic_tool("read_file", ToolStatus::Success);
479 cell.input_summary = Some("command: git diff --stat".to_string());
480 cell.output = Some(diff_stat.to_string());
481 HistoryCell::Tool(ToolCell::Generic(cell))
482 };
483
484 let transcript_text = lines_text(&cell.transcript_lines(80));
485 for needle in [
486 "Cargo.lock",
487 "crates/cli/Cargo.toml",
488 "crates/cli/src/main.rs",
489 "crates/config/src/lib.rs",
490 "crates/tui/src/mcp.rs",
491 ] {
492 assert!(
493 transcript_text.contains(needle),
494 "transcript missing {needle:?}: {transcript_text}"
495 );
496 }
497 let cargo_lock_row = transcript_text
498 .lines()
499 .find(|line| line.contains("Cargo.lock"))
500 .expect("Cargo.lock row must exist");
501 assert!(
502 !cargo_lock_row.contains("crates/cli/Cargo.toml"),
503 "two files were joined onto one row: {cargo_lock_row}"
504 );
505 }
506
507 /// Reasoning folds in the live view and the fold is reversible: the collapsed
508 /// state must truncate a long body, the expanded state must restore every line
509 /// it dropped, and both must show the model's own identifiers verbatim (the
510 /// #4146/#4148 scrub rendered `refresh_catalog_cache` as `…` and protected
511 /// nothing, since the body was always one keypress away). The configured
512 /// default only inverts which state the toggle starts in.
513 ///
514 /// Replaces four separate fold tests.
515 #[test]
516 fn reasoning_folds_in_live_and_the_fold_is_reversible() {
517 let body = (1..=20)
518 .map(|i| format!("step {i:02}: refresh_catalog_cache iteration"))
519 .collect::<Vec<_>>()
520 .join("\n");
521 let cell = HistoryCell::Thinking {
522 content: body,
523 streaming: false,
524 duration_secs: Some(1.0),
525 };
526
527 for default_expanded in [false, true] {
528 let options = TranscriptRenderOptions {
529 thinking_default_expanded: default_expanded,
530 low_motion: true,
531 ..TranscriptRenderOptions::default()
532 };
533 // An explicit intent says expanded or collapsed outright, so the same
534 // two calls answer for either configured default. Running both proves
535 // the intent is not re-read through the preference.
536 let expanded = lines_text(
537 &cell
538 .lines_with_options_folded(80, options, Some(ThinkingFold::Expanded))
539 .0,
540 );
541 let collapsed = lines_text(
542 &cell
543 .lines_with_options_folded(80, options, Some(ThinkingFold::Collapsed))
544 .0,
545 );
546
547 for i in 1..=20 {
548 assert!(
549 expanded.contains(&format!("step {i:02}: refresh_catalog_cache iteration")),
550 "[default_expanded={default_expanded}] expanded reasoning dropped line {i}: \
551 {expanded}"
552 );
553 }
554 assert!(
555 !collapsed.contains("step 20:"),
556 "[default_expanded={default_expanded}] the collapsed fold must truncate: {collapsed}"
557 );
558 assert!(
559 collapsed.contains("refresh_catalog_cache"),
560 "[default_expanded={default_expanded}] the shown head keeps identifiers \
561 verbatim: {collapsed}"
562 );
563 assert!(
564 !collapsed.contains("Space:") && !collapsed.contains("Ctrl+O"),
565 "[default_expanded={default_expanded}] the per-cell renderer stays \
566 target-neutral; the chord belongs to whoever owns focus: {collapsed}"
567 );
568 }
569 }
570
571 /// The preference baseline (verbose session or expanded default) decides only
572 /// the cells nobody has touched; an explicit intent decides its own cell in
573 /// both directions. Verbose plus an expanded default renders expanded — the
574 /// old triple-XOR collapsed exactly that cell — and an explicit expand stays
575 /// expanded under every preference combination, which the later relative bit
576 /// still got wrong (#5847).
577 #[test]
578 fn explicit_thinking_fold_outranks_every_preference_baseline() {
579 let body = (1..=20)
580 .map(|i| format!("step {i:02}: baseline check"))
581 .collect::<Vec<_>>()
582 .join("\n");
583 let cell = HistoryCell::Thinking {
584 content: body,
585 streaming: false,
586 duration_secs: Some(1.0),
587 };
588 // (fold, verbose, default_expanded, expect_expanded)
589 for (fold, verbose, default_expanded, expect_expanded) in [
590 // No explicit intent: the preference baseline decides.
591 (None, false, false, false),
592 (None, false, true, true),
593 (None, true, false, true),
594 (None, true, true, true),
595 // An explicit expand renders expanded whatever the preferences say.
596 (Some(ThinkingFold::Expanded), false, false, true),
597 (Some(ThinkingFold::Expanded), false, true, true),
598 (Some(ThinkingFold::Expanded), true, false, true),
599 (Some(ThinkingFold::Expanded), true, true, true),
600 // And an explicit collapse renders collapsed whatever they say.
601 (Some(ThinkingFold::Collapsed), false, false, false),
602 (Some(ThinkingFold::Collapsed), false, true, false),
603 (Some(ThinkingFold::Collapsed), true, false, false),
604 (Some(ThinkingFold::Collapsed), true, true, false),
605 ] {
606 let options = TranscriptRenderOptions {
607 verbose,
608 thinking_default_expanded: default_expanded,
609 low_motion: true,
610 ..TranscriptRenderOptions::default()
611 };
612 let text = lines_text(&cell.lines_with_options_folded(80, options, fold).0);
613 let expanded = text.contains("step 20: baseline check");
614 assert_eq!(
615 expanded, expect_expanded,
616 "[fold={fold:?} verbose={verbose} default_expanded={default_expanded}]"
617 );
618 }
619 }
620
621 /// A completed reasoning cell short enough to fit needs no expand affordance,
622 /// and the live view must still show it — the alternative was a dead card that
623 /// said reasoning happened and nothing about what it was.
624 #[test]
625 fn short_completed_reasoning_is_shown_live_without_an_affordance() {
626 let cell = HistoryCell::Thinking {
627 content: "One brief reasoning step.".to_string(),
628 streaming: false,
629 duration_secs: Some(0.4),
630 };
631
632 let live_text = lines_text(&cell.lines_with_options(80, calm_options()));
633 let transcript_text = lines_text(&cell.transcript_lines(80));
634
635 assert!(
636 live_text.contains("One brief reasoning step."),
637 "short completed reasoning belongs inline: {live_text}"
638 );
639 assert!(transcript_text.contains("One brief reasoning step."));
640 assert!(
641 !live_text.contains("Ctrl+O") && !live_text.contains("Space:"),
642 "a body that fits needs no affordance: {live_text}"
643 );
644 }
645
646 /// A live reasoning block must show what the model is thinking right now — the
647 /// old behavior stalled on a `thinking...` placeholder until the block closed,
648 /// and a long body must keep the newest line rather than the oldest.
649 #[test]
650 fn streaming_reasoning_shows_its_newest_line_not_a_placeholder() {
651 let short = render_thinking(
652 "Step 1: read the code\nStep 2: trace the call\nStep 3: form a hypothesis",
653 80,
654 true,
655 None,
656 true,
657 true,
658 );
659 let short_text = lines_text(&short);
660 assert!(
661 short_text.contains("Step 3: form a hypothesis"),
662 "the newest reasoning line must be visible while streaming: {short_text}"
663 );
664 assert!(
665 !short_text.contains("thinking..."),
666 "real content means the placeholder must not be drawn: {short_text}"
667 );
668
669 let long = (1..=16)
670 .map(|i| format!("Reasoning line {i}"))
671 .collect::<Vec<_>>()
672 .join("\n");
673 let long_text = lines_text(&render_thinking(&long, 80, true, None, true, true));
674 assert!(
675 long_text.contains("Reasoning line 16"),
676 "the tail is what is live: {long_text}"
677 );
678 assert!(
679 !long_text.contains("Reasoning line 1\n"),
680 "the head is what gets clipped: {long_text}"
681 );
682 }
683
684 /// A foreground shell wait blocks the turn. The card's job is to tell the user
685 /// how to take the terminal back, not to re-print the command they just watched
686 /// the model type, and not to duplicate the sidebar's live tail in the
687 /// transcript. Once the command finishes, the final output supersedes any stale
688 /// live tail.
689 ///
690 /// Replaces three tests.
691 #[test]
692 fn a_foreground_shell_wait_offers_the_escape_hatch_not_the_command_echo() {
693 let command = "cargo test --workspace --all-features";
694 let running = {
695 let mut exec = exec_tool(command, ToolStatus::Running);
696 exec.live_output = Some("running line 1\nrunning line 2".to_string());
697 exec.shell_task_id = Some("shell_live".to_string());
698 exec
699 };
700
701 for (label, text) in [
702 ("live", lines_text(&running.lines_with_motion(80, true))),
703 (
704 "transcript",
705 lines_text(&HistoryCell::Tool(ToolCell::Exec(running.clone())).transcript_lines(80)),
706 ),
707 ] {
708 assert!(
709 text.contains("Ctrl+B"),
710 "[{label}] the backgrounding chord is the point of the card: {text}"
711 );
712 assert!(
713 !text.contains("running line 1"),
714 "[{label}] the live tail belongs to the sidebar and /jobs: {text}"
715 );
716 assert!(
717 !text.contains(command),
718 "[{label}] the header already carries the summary; do not echo the \
719 command target: {text}"
720 );
721 assert!(!text.contains("command:"), "[{label}] {text}");
722 }
723
724 let mut finished = exec_tool(command, ToolStatus::Success);
725 finished.output = Some("final output".to_string());
726 finished.live_output = Some("stale live tail".to_string());
727 finished.shell_task_id = Some("shell_live".to_string());
728 let text = lines_text(&finished.lines_with_motion(80, true));
729 assert!(
730 !text.contains("stale live tail"),
731 "a finished command must not show the tail it already superseded: {text}"
732 );
733 }
734
735 // ---------------------------------------------------------------------------
736 // Clipboard — what you copy is what was authored
737 // ---------------------------------------------------------------------------
738
739 /// Every rendered line carries a `copy_prefix_width`: the display columns of
740 /// decoration the clipboard must skip. The property is that slicing a line at
741 /// that width yields the payload and nothing decorative — for role markers,
742 /// status chrome, and the two-column continuation prefix on wrapped fenced code
743 /// (which must be counted in display columns, not bytes, or CJK shifts it).
744 ///
745 /// Replaces three tests that each covered one cell kind.
746 #[test]
747 fn the_copy_prefix_skips_every_decoration_and_keeps_the_payload() {
748 let decorations = ['╎', '▎', '●', '│', '┃', '✓', '▏'];
749
750 let copied_line = |cell: &HistoryCell, width: u16, needle: &str| -> (String, usize) {
751 let rendered = cell.lines_with_copy_metadata(width, TranscriptRenderOptions::default());
752 let target = rendered
753 .iter()
754 .find(|entry| {
755 entry
756 .line
757 .spans
758 .iter()
759 .any(|span| span.content.contains(needle))
760 })
761 .unwrap_or_else(|| panic!("no rendered line contains {needle:?}"));
762 let text = line_to_plain(&target.line);
763 (
764 slice_visible_columns(&text, target.copy_prefix_width, text_visible_width(&text)),
765 target.copy_prefix_width,
766 )
767 };
768
769 // Fenced code: indentation survives, decoration does not.
770 let rust_fence = HistoryCell::Assistant {
771 content: "```rust\n let answer = 42;\n```".to_string(),
772 streaming: false,
773 };
774 let (copied, _) = copied_line(&rust_fence, 40, "answer");
775 assert!(
776 copied.contains(" let answer = 42;"),
777 "code indentation was not preserved: {copied:?}"
778 );
779 for glyph in decorations {
780 assert!(
781 !copied.contains(glyph),
782 "decorative glyph {glyph:?} leaked into copied code: {copied:?}"
783 );
784 }
785
786 // Wrapped CJK code: the prefix is two *display* columns, not two bytes.
787 let cjk_fence = HistoryCell::Assistant {
788 content: "```text\n 中文 = 1\n```".to_string(),
789 streaming: false,
790 };
791 let (copied, prefix) = copied_line(&cjk_fence, 24, "中文");
792 assert_eq!(
793 prefix, 2,
794 "the continuation prefix is the role marker's two display columns"
795 );
796 assert!(
797 copied.starts_with(" 中文"),
798 "wide-character indentation was mis-sliced: {copied:?}"
799 );
800
801 // Tool receipt: status and family chrome are prefix, the receipt text is
802 // payload.
803 let receipt = {
804 let mut exec = exec_tool("printf 'receipt'", ToolStatus::Success);
805 exec.output = Some("receipt".to_string());
806 HistoryCell::Tool(ToolCell::Exec(exec))
807 };
808 let rendered = receipt.lines_with_copy_metadata(80, TranscriptRenderOptions::default());
809 let header = rendered.first().expect("tool receipt header");
810 assert!(
811 header.copy_prefix_width >= 4,
812 "status and family chrome should be measured as prefix, got {}",
813 header.copy_prefix_width
814 );
815 let body = line_to_plain(&ratatui::text::Line::from(
816 header
817 .line
818 .spans
819 .iter()
820 .skip(1)
821 .cloned()
822 .collect::<Vec<_>>(),
823 ));
824 let copied = slice_visible_columns(&body, header.copy_prefix_width, text_visible_width(&body));
825 assert!(
826 copied.contains("run Done"),
827 "receipt text was clipped away: {copied:?}"
828 );
829 for glyph in decorations {
830 assert!(
831 !copied.contains(glyph),
832 "decorative glyph {glyph:?} leaked into the copied receipt: {copied:?}"
833 );
834 }
835 }
836
837 /// Issue #1212: the transcript rail (`▏`) marks prose continuation. Inside a
838 /// fence it corrupts anything the user copies, so no line of a code block may
839 /// carry it — not the first, not a blank line in the middle, not a wrapped
840 /// continuation of an over-long source line.
841 ///
842 /// Replaces four tests that each covered one fence shape.
843 #[test]
844 fn no_line_inside_a_fence_carries_the_transcript_rail() {
845 let long_source = "let x = ".to_string() + &"abcdef ".repeat(40);
846
847 for (label, content, width) in [
848 (
849 "short fence",
850 "SQL:\n```sql\nSELECT\nFROM customers\n```".to_string(),
851 80u16,
852 ),
853 (
854 "multi-line fence",
855 "Here's the query:\n```sql\nSELECT\n c.customer_id,\n c.name,\n \
856 COUNT(o.order_id) AS order_count\nFROM customers c\nJOIN orders o ON \
857 c.customer_id = o.customer_id;\n```"
858 .to_string(),
859 80,
860 ),
861 (
862 "fence with a blank line",
863 "```\nfn one() {}\n\nfn two() {}\n```".to_string(),
864 80,
865 ),
866 ("wrapped fence", format!("```\n{long_source}\n```"), 40),
867 ] {
868 let cell = HistoryCell::Assistant {
869 content,
870 streaming: false,
871 };
872 // Line 0 is the intro paragraph (or the fence opener); every line
873 // after it belongs to the code block.
874 for line in cell.lines(width).iter().skip(1) {
875 let text = line_text(line);
876 assert!(
877 !text.contains('\u{258F}'),
878 "[{label}] code line took the transcript rail: {text:?}"
879 );
880 }
881 }
882 }
883
884 /// Whose text gets interpreted is a trust boundary. The model's markdown is
885 /// rendered; the user's prompt is shown exactly as typed, including leading
886 /// hashes, dashes and runs of spaces. A cell holding only whitespace renders
887 /// nothing at all rather than an orphaned role glyph.
888 ///
889 /// Replaces three tests.
890 #[test]
891 fn authored_text_keeps_its_shape_on_both_sides_of_the_turn() {
892 let user = HistoryCell::User {
893 content: " # heading\n- item\n \nhello world".to_string(),
894 };
895 let visible: Vec<String> = user.lines(80).iter().map(line_text).collect();
896 assert!(
897 visible[0].trim_end().ends_with("# heading"),
898 "a user's literal `#` must not become a rendered heading: {visible:?}"
899 );
900 assert!(
901 visible[1].trim_end().ends_with("- item"),
902 "dash-prefixed user text stays literal: {visible:?}"
903 );
904 assert!(
905 visible[2].ends_with(" "),
906 "whitespace-only user lines survive: {visible:?}"
907 );
908 assert!(
909 visible[3].trim_end().ends_with("hello world"),
910 "internal spacing stays literal: {visible:?}"
911 );
912 assert!(
913 !visible.iter().any(|line| line.contains('\u{2500}')),
914 "user text must not gain a markdown heading rule: {visible:?}"
915 );
916
917 let assistant = HistoryCell::Assistant {
918 content: "# Heading\n\n- item".to_string(),
919 streaming: false,
920 };
921 let visible: Vec<String> = assistant.lines(80).iter().map(line_text).collect();
922 assert!(
923 visible[0].contains("Heading") && !visible[0].contains("# Heading"),
924 "the model's markdown is still parsed: {visible:?}"
925 );
926 assert!(
927 visible.iter().any(|line| line.contains('\u{2500}')),
928 "an assistant h1 still draws its rule: {visible:?}"
929 );
930
931 // A stray newline streamed between reasoning and a tool call used to render
932 // as a bare role glyph with nothing after it.
933 for content in ["", " ", "\n", "\n\n", " \t \n"] {
934 for streaming in [false, true] {
935 let cell = HistoryCell::Assistant {
936 content: content.to_string(),
937 streaming,
938 };
939 assert!(
940 cell.lines(80).is_empty(),
941 "whitespace-only assistant content {content:?} (streaming={streaming}) \
942 must render nothing"
943 );
944 }
945 }
946 let real = HistoryCell::Assistant {
947 content: "hi".to_string(),
948 streaming: false,
949 };
950 assert_eq!(
951 real.lines(80)[0].spans[0].content.as_ref(),
952 ASSISTANT_GLYPH,
953 "real content still gets its role marker"
954 );
955 }
956
957 /// Reasoning is neither the user's prompt nor the model's answer, and a reader
958 /// scanning the transcript has to be able to skip it. The markers below are
959 /// referenced as named constants rather than literal glyphs on purpose: this
960 /// protects the *distinction*, so a redesign that restyles reasoning stays
961 /// green while one that stops marking it at all fails.
962 #[test]
963 fn reasoning_is_marked_apart_from_both_the_prompt_and_the_answer() {
964 let body_text = "concrete reasoning content";
965 let reasoning = render_thinking(body_text, 80, false, Some(1.0), false, true);
966 assert!(reasoning.len() >= 2, "expected a header and a body line");
967
968 let header = line_text(&reasoning[0]);
969 assert!(
970 header.starts_with(REASONING_OPENER),
971 "the reasoning header opens with its own marker: {header:?}"
972 );
973 let body = line_text(&reasoning[1]);
974 assert!(
975 body.starts_with(REASONING_RAIL),
976 "the reasoning body carries its own rail: {body:?}"
977 );
978
979 let rail = REASONING_RAIL.trim();
980 for cell in [
981 HistoryCell::User {
982 content: body_text.to_string(),
983 },
984 HistoryCell::Assistant {
985 content: body_text.to_string(),
986 streaming: false,
987 },
988 ] {
989 let rendered = lines_text(&cell.lines(80));
990 assert!(
991 rendered.contains(body_text),
992 "sanity: the cell rendered its content: {rendered}"
993 );
994 assert!(
995 !rendered.contains(rail),
996 "only reasoning may wear the reasoning rail: {rendered}"
997 );
998 }
999 }
1000
1001 /// A filled background behind reasoning is unreadable on a transparent or
1002 /// light terminal, so the highlight is configurable. With it off, not one span
1003 /// may carry a background — a single tinted span is the bug. The enabled case
1004 /// follows the terminal's actual color depth: capable terminals tint the body,
1005 /// while ANSI-16 intentionally stays untinted because it cannot render the
1006 /// subtle surface faithfully.
1007 #[test]
1008 fn disabling_the_reasoning_highlight_leaves_no_span_with_a_background() {
1009 let render = |highlight: bool| {
1010 render_thinking_with_analysis(
1011 "reasoning without a filled surface",
1012 80,
1013 false,
1014 Some(1.0),
1015 false,
1016 true,
1017 highlight,
1018 )
1019 .0
1020 };
1021
1022 assert!(
1023 render(false)
1024 .iter()
1025 .flat_map(|line| line.spans.iter())
1026 .all(|span| span.style.bg.is_none()),
1027 "a disabled highlight must not tint any span"
1028 );
1029 let enabled_has_background = render(true)
1030 .iter()
1031 .flat_map(|line| line.spans.iter())
1032 .any(|span| span.style.bg.is_some());
1033 assert_eq!(
1034 enabled_has_background,
1035 codewhale_palette::reasoning_surface_tint(cached_color_depth()).is_some(),
1036 "the enabled highlight must follow the terminal color-depth contract"
1037 );
1038 }
1039
1040 // ---------------------------------------------------------------------------
1041 // Motion — reduced motion is actually still
1042 // ---------------------------------------------------------------------------
1043
1044 /// The deleted tests pinned the frozen glyphs (`assert_eq!(spans[1], "⣤")`).
1045 /// That breaks on a skin change and passes on the bug that matters: a marker
1046 /// that keeps animating for a user who asked it to stop. The property is
1047 /// stillness — a running cell's rendered frame must not depend on how long it
1048 /// has been running once motion is reduced — and the full-motion case is
1049 /// asserted alongside it so a renderer that froze everything could not make
1050 /// this test vacuously true.
1051 ///
1052 /// Stillness is not enough on its own. Animation frame 0 is U+2800 BRAILLE
1053 /// PATTERN BLANK, an invisible cell. Freezing there (or on the Still path)
1054 /// looks like a missing marker, which is why reduced motion must freeze on a
1055 /// filled, legible bubble rather than the blank the spinner starts on.
1056 #[test]
1057 fn reduced_and_still_motion_render_a_frame_that_does_not_move() {
1058 let frame_symbols = super::TOOL_RUNNING_SYMBOLS.len() as u64;
1059 let frame_at = |elapsed_ms: u64, low_motion: bool, motion: MotionMode| {
1060 let mut exec = exec_tool("echo hi", ToolStatus::Running);
1061 exec.started_at = Some(Instant::now() - Duration::from_millis(elapsed_ms));
1062 let cell = HistoryCell::Tool(ToolCell::Exec(exec));
1063 lines_text(&cell.lines_with_options(
1064 80,
1065 TranscriptRenderOptions {
1066 low_motion,
1067 motion_mode: motion,
1068 ..TranscriptRenderOptions::default()
1069 },
1070 ))
1071 };
1072
1073 // Half a spinner cycle apart, and both well under the 3s elapsed-badge
1074 // threshold so the badge itself cannot be the thing that differs.
1075 let early = crate::tui::spinner::LIVE_MARKER_DELAY_MS;
1076 let late = early + super::TOOL_STATUS_SYMBOL_MS * (frame_symbols / 2);
1077 assert!(
1078 late < 3_000,
1079 "both samples must stay under the elapsed badge"
1080 );
1081
1082 // Two independent mechanisms are supposed to produce stillness — the
1083 // `low_motion` flag and the resolved `motion_mode`. Each is asserted on its
1084 // own so losing either one fails here, rather than only losing both.
1085 for (low_motion, motion) in [
1086 (true, MotionMode::Reduced),
1087 (true, MotionMode::Still),
1088 (false, MotionMode::Reduced),
1089 (false, MotionMode::Still),
1090 ] {
1091 let frozen = frame_at(early, low_motion, motion);
1092 assert_eq!(
1093 frozen,
1094 frame_at(late, low_motion, motion),
1095 "low_motion={low_motion} / {motion:?} must not animate the live marker"
1096 );
1097 assert!(
1098 !frozen.contains('\u{2800}'),
1099 "a frozen marker must still be visible: low_motion={low_motion} / {motion:?}: {frozen:?}"
1100 );
1101 }
1102 assert_ne!(
1103 frame_at(early, false, MotionMode::Full),
1104 frame_at(late, false, MotionMode::Full),
1105 "full motion must actually animate, or the stillness assertions above \
1106 prove nothing"
1107 );
1108
1109 // The same contract for the two other animated surfaces: the streaming
1110 // reasoning cursor and the assistant role marker's pulse.
1111 let cursor_off = lines_text(&render_thinking(
1112 "ongoing reasoning...",
1113 80,
1114 true,
1115 None,
1116 false,
1117 true,
1118 ));
1119 assert!(
1120 !cursor_off.contains(REASONING_CURSOR),
1121 "low motion must suppress the streaming reasoning cursor: {cursor_off}"
1122 );
1123 assert_eq!(
1124 assistant_label_style_for(true, true).fg,
1125 assistant_label_style_for(false, false).fg,
1126 "a streaming assistant marker under low motion must look exactly like an \
1127 idle one — no pulse"
1128 );
1129 }
1130
1131 /// Dual of the low-motion freeze above: when the cell is streaming and
1132 /// motion is allowed, the assistant marker must actually pulse. The deleted
1133 /// test slept up to 1s sampling `SystemTime` until the 2s sine dipped; the
1134 /// property is that the streaming+motion color is `pulse_brightness` of the
1135 /// idle source, which we can check without waiting on the wall clock.
1136 ///
1137 /// Around the sine crest, `pulse_brightness` rounds back to the source
1138 /// (~70ms of a 2s cycle). Matching the current instant would then also pass
1139 /// a renderer that never pulsed, so we only compare once the pure function
1140 /// itself is off the crest — a busy wait, not a sleep.
1141 #[test]
1142 fn assistant_marker_pulses_when_streaming_and_motion_is_allowed() {
1143 use codewhale_palette::{self as palette, pulse_brightness};
1144
1145 let idle = assistant_label_style_for(false, false).fg;
1146 assert_eq!(
1147 idle,
1148 Some(palette::WHALE_ACTION),
1149 "the idle marker is the unpulsed source; pulsing everything would make \
1150 the streaming assertion vacuously true"
1151 );
1152 assert_eq!(
1153 assistant_label_style_for(true, true).fg,
1154 idle,
1155 "low motion must keep the streaming marker at the unpulsed source"
1156 );
1157
1158 let epoch_ms = || {
1159 std::time::SystemTime::now()
1160 .duration_since(std::time::UNIX_EPOCH)
1161 .map(|d| d.as_millis() as u64)
1162 .unwrap_or(0)
1163 };
1164 let deadline = Instant::now() + Duration::from_millis(250);
1165 let (t0, actual) = loop {
1166 assert!(
1167 Instant::now() < deadline,
1168 "pulse_brightness stayed at the source color through a 250ms spin; \
1169 the 2s cycle leaves the crest in ~70ms"
1170 );
1171 let t0 = epoch_ms();
1172 // Skip the crest and a few ms of margin so the product read of
1173 // SystemTime cannot land back on identity between this sample and
1174 // the call under test.
1175 let near_crest = (t0.saturating_sub(8)..=t0.saturating_add(8))
1176 .any(|ms| pulse_brightness(palette::WHALE_ACTION, ms) == palette::WHALE_ACTION);
1177 if near_crest {
1178 continue;
1179 }
1180 break (t0, assistant_label_style_for(true, false).fg);
1181 };
1182 let t1 = epoch_ms();
1183 let matches_pulse =
1184 (t0..=t1.max(t0)).any(|ms| actual == Some(pulse_brightness(palette::WHALE_ACTION, ms)));
1185 assert!(
1186 matches_pulse,
1187 "streaming + motion must apply pulse_brightness to the assistant \
1188 marker, got {actual:?}"
1189 );
1190 assert_ne!(
1191 actual, idle,
1192 "streaming + motion must not sit at the idle color once the pulse \
1193 is off its crest"
1194 );
1195 }
1196
1197 /// The still-motion path rewrites the leading status marker in place. It once
1198 /// rewrote any braille cell it found, which silently ate braille that was part
1199 /// of the tool's own output.
1200 #[test]
1201 fn the_still_marker_rewrite_never_consumes_braille_tool_output() {
1202 let mut cell = generic_tool("read_file", ToolStatus::Running);
1203 cell.output = Some("⣿".to_string());
1204 let cell = HistoryCell::Tool(ToolCell::Generic(cell));
1205
1206 let lines = cell.lines_with_options(
1207 80,
1208 TranscriptRenderOptions {
1209 low_motion: true,
1210 motion_mode: MotionMode::Still,
1211 ..TranscriptRenderOptions::default()
1212 },
1213 );
1214
1215 assert!(
1216 lines
1217 .iter()
1218 .flat_map(|line| line.spans.iter())
1219 .any(|span| span.content.as_ref() == "⣿"),
1220 "tool output must survive the typed-header marker pass: {lines:?}"
1221 );
1222 }
1223
1224 // ---------------------------------------------------------------------------
1225 // Identity — a card never names a verb or a tool it did not run
1226 // ---------------------------------------------------------------------------
1227
1228 /// #4145: a completed grep grouped under the exploration card rendered
1229 /// `read done · Searching …` — the header verb contradicted the label directly
1230 /// under it. The verb must agree with the work, in every locale, and a locale
1231 /// must not fall back to the English status word.
1232 ///
1233 /// Replaces two tests, each of which hard-coded one direction.
1234 #[test]
1235 fn a_card_verb_agrees_with_its_own_label_in_every_locale() {
1236 use codewhale_localization::Locale;
1237
1238 for (label, expected_en, expected_zh, forbidden_en) in [
1239 (
1240 "Searching for `TranscriptScroll`",
1241 "find Done",
1242 "find 完成",
1243 "read Done",
1244 ),
1245 ("Reading src/foo.rs", "read Done", "read 完成", "find Done"),
1246 ] {
1247 let cell = super::ExploringCell {
1248 entries: vec![super::ExploringEntry {
1249 label: label.to_string(),
1250 status: ToolStatus::Success,
1251 }],
1252 };
1253
1254 let header_en = line_text(&cell.lines_with_motion(80, true)[0]);
1255 assert!(
1256 header_en.contains(expected_en),
1257 "{label:?} should read {expected_en:?}: {header_en:?}"
1258 );
1259 assert!(
1260 !header_en.contains(forbidden_en),
1261 "{label:?} must not be paired with {forbidden_en:?}: {header_en:?}"
1262 );
1263 assert!(
1264 header_en.contains(label),
1265 "the label itself must survive: {header_en:?}"
1266 );
1267
1268 let header_zh = line_text(&cell.lines_with_motion_and_locale(80, true, Locale::ZhHans)[0]);
1269 assert!(
1270 header_zh.contains(expected_zh),
1271 "{label:?} should read {expected_zh:?} in zh-Hans: {header_zh:?}"
1272 );
1273 assert!(
1274 !header_zh.to_lowercase().contains("done"),
1275 "zh-Hans must not leak the English status word: {header_zh:?}"
1276 );
1277 assert!(
1278 header_zh.contains(label),
1279 "the label itself must survive localization: {header_zh:?}"
1280 );
1281 }
1282 }
1283
1284 /// A read/find receipt reports a line count, so the count has to be real —
1285 /// including the singular/plural and the localized unit. A run receipt reports
1286 /// no count at all: inferring "3 lines" from rendered text that happens to
1287 /// contain `stdout:` would be inventing a number the shell never reported.
1288 #[test]
1289 fn receipts_count_only_what_they_actually_counted() {
1290 use crate::tui::widgets::tool_card::ToolFamily;
1291 use codewhale_localization::Locale;
1292
1293 for (locale, done, unit) in [(Locale::En, "Done", "line"), (Locale::ZhHans, "完成", "行")] {
1294 let label = |family, status, output| {
1295 super::tool_receipt_label(family, status, Some(output), locale)
1296 };
1297
1298 assert_eq!(label(ToolFamily::Read, ToolStatus::Success, ""), done);
1299 assert_eq!(
1300 label(ToolFamily::Read, ToolStatus::Success, "hello\n"),
1301 if locale == Locale::En {
1302 "1 line".to_string()
1303 } else {
1304 format!("1 {unit}")
1305 }
1306 );
1307 assert_eq!(
1308 label(ToolFamily::Read, ToolStatus::Success, "a\nb\nc\n"),
1309 if locale == Locale::En {
1310 "3 lines".to_string()
1311 } else {
1312 format!("3 {unit}")
1313 }
1314 );
1315 assert_eq!(
1316 label(ToolFamily::Find, ToolStatus::Success, "match 1\nmatch 2\n"),
1317 if locale == Locale::En {
1318 "2 lines".to_string()
1319 } else {
1320 format!("2 {unit}")
1321 }
1322 );
1323
1324 // Run never counts, whatever the body looks like.
1325 for body in [
1326 "stdout:\nok\nmore\nstderr:\nbad\n",
1327 "line 1\nline 2\nline 3\n",
1328 ] {
1329 assert_eq!(
1330 label(ToolFamily::Run, ToolStatus::Success, body),
1331 done,
1332 "a run receipt must not infer counts from {body:?}"
1333 );
1334 }
1335 }
1336
1337 assert_eq!(
1338 super::tool_receipt_label(
1339 ToolFamily::Read,
1340 ToolStatus::Running,
1341 Some("a\nb"),
1342 Locale::En
1343 ),
1344 "running",
1345 "an unfinished read has nothing to count yet"
1346 );
1347 }
1348
1349 /// The same truthfulness contract through the real shell render path, where a
1350 /// formatter has already rewritten the output: the header still reports a plain
1351 /// localized completion and never a fabricated line count or stream name.
1352 #[test]
1353 fn shell_headers_stay_truthful_through_the_output_formatters() {
1354 use codewhale_localization::Locale;
1355
1356 let cases = [
1357 (
1358 "printf redirect",
1359 "printf '%s\\n' 'hello' 'world' > src/main.rs",
1360 "printf > src/main.rs\nhello\nworld\n",
1361 ),
1362 (
1363 "logical-or fallback",
1364 "cargo build || echo fallback",
1365 " Compiling pkg v0.1.0\n Finished dev [unoptimized + debuginfo]\n",
1366 ),
1367 ];
1368
1369 for (label, command, output) in cases {
1370 let mut cell = exec_tool(command, ToolStatus::Success);
1371 cell.output = Some(output.to_string());
1372 cell.duration_ms = Some(42);
1373
1374 for (locale, done, unit) in [
1375 (Locale::En, "Done", "lines"),
1376 (Locale::ZhHans, "完成", "行"),
1377 ] {
1378 let header = line_text(&cell.render_with_locale(80, true, RenderMode::Live, locale)[0]);
1379 assert!(
1380 header.contains(done),
1381 "[{label}] header must carry the localized completion: {header}"
1382 );
1383 assert!(
1384 !header.contains(unit) && !header.contains("stdout") && !header.contains("stderr"),
1385 "[{label}] header must not invent counts or stream names: {header}"
1386 );
1387 }
1388 }
1389 }
1390
1391 /// #4133 / #4148: a spawn yields its card entirely to the DelegateCard, and an
1392 /// inspection (`peek` / `wait` / `status`) is a one-line check in every render
1393 /// mode. It must name the child it checked, must not read as a completed
1394 /// delegation, and must not leak the internal "unknown child" placeholder or
1395 /// echo the verb twice when the resolved identity collapses onto it.
1396 ///
1397 /// Replaces eight tests.
1398 #[test]
1399 fn agent_cards_stay_one_line_and_spawn_cards_yield_to_the_delegate_card() {
1400 let agent = |summary: &str, output: Option<&str>| {
1401 let mut cell = generic_tool("agent", ToolStatus::Success);
1402 cell.input_summary = Some(summary.to_string());
1403 cell.output = output.map(str::to_string);
1404 cell
1405 };
1406
1407 for mode in [RenderMode::Live, RenderMode::Transcript] {
1408 let spawn = agent(
1409 "prompt: map the repo",
1410 Some(r#"{"agent_id":"agent_scout_1","status":"running"}"#),
1411 );
1412 assert!(
1413 spawn.lines_with_mode(120, true, mode).is_empty(),
1414 "a spawn must not draw a generic card beside the DelegateCard in {mode:?}"
1415 );
1416
1417 for (summary, output, expected) in [
1418 (
1419 "action: peek agent_id: agent_scout_1",
1420 Some(r#"{"agent_id":"agent_scout_1","status":"running"}"#),
1421 "checked",
1422 ),
1423 (
1424 "action: wait",
1425 Some(r#"{"action":"wait","settled":[{"agent_id":"agent_scout_1"}]}"#),
1426 "waited",
1427 ),
1428 (
1429 "action: status agent_id: agent_scout_1",
1430 Some(r#"{"agent_id":"agent_scout_1","status":"running","terminal":false}"#),
1431 "checked",
1432 ),
1433 ] {
1434 let cell = agent(summary, output);
1435 let lines = cell.lines_with_mode(120, true, mode);
1436 let text = lines_text(&lines);
1437 assert_eq!(
1438 lines.len(),
1439 1,
1440 "{summary:?} must stay one line in {mode:?}: {lines:?}"
1441 );
1442 assert!(
1443 text.contains(expected),
1444 "{summary:?} should read as {expected:?}: {text:?}"
1445 );
1446 assert!(
1447 !text.to_lowercase().contains("delegate done"),
1448 "an inspection must not read as a finished delegation: {text:?}"
1449 );
1450 }
1451 }
1452
1453 // Identity fallbacks: no raw placeholder, no doubled verb.
1454 let unresolved = agent("action: peek agent_type: delegate", None);
1455 let text = lines_text(&unresolved.lines_with_mode(80, true, RenderMode::Live));
1456 assert!(
1457 !text.contains("unknown child"),
1458 "the internal fallback token must not reach the transcript: {text:?}"
1459 );
1460
1461 let collapsing = agent("action: peek role: delegate", None);
1462 let text = lines_text(&collapsing.lines_with_mode(80, true, RenderMode::Live));
1463 assert!(
1464 text.contains(" agent "),
1465 "the agent family label should appear once in the summary: {text:?}"
1466 );
1467 }
1468
1469 /// A tool the catalog does not have produces one useful sentence — the catalog
1470 /// error — and nothing else. The old rendering spent a `name:` / `args:` /
1471 /// `result:` block restating a call that never happened.
1472 #[test]
1473 fn an_unknown_tool_failure_shows_only_the_catalog_error() {
1474 let mut cell = generic_tool("item", ToolStatus::Failed);
1475 cell.input_summary = Some("status: pending".to_string());
1476 cell.output = Some(
1477 "Tool 'item' is not available in the current tool catalog. \
1478 Checklist entries are not separate tool calls."
1479 .to_string(),
1480 );
1481
1482 for mode in [RenderMode::Live, RenderMode::Transcript] {
1483 let lines = cell.lines_with_mode(120, true, mode);
1484 let text = lines_text(&lines);
1485 assert_eq!(lines.len(), 1, "single header line in {mode:?}: {lines:?}");
1486 assert!(
1487 text.contains("Tool 'item' is not available"),
1488 "the catalog error is the useful part: {text:?}"
1489 );
1490 assert!(
1491 !text.contains("name: item"),
1492 "no name/args/result block for a call that did not happen: {text:?}"
1493 );
1494 }
1495 }
1496
1497 // ---------------------------------------------------------------------------
1498 // Severity — the ranks stay distinguishable
1499 // ---------------------------------------------------------------------------
1500
1501 /// The deleted tests pinned each severity to a named palette constant, so a
1502 /// theme change broke four tests and a severity collapse broke none. What has
1503 /// to hold is the *relationship*: `Critical` reads exactly as loud as `Error`,
1504 /// `Warning` is visibly not an error, and `Info` is quieter than both — so a
1505 /// transient retry cannot be mistaken for a hard failure sitting next to it.
1506 #[test]
1507 fn error_severity_ranks_stay_visually_distinguishable() {
1508 use crate::error_taxonomy::ErrorSeverity;
1509
1510 let rank = |severity| {
1511 let cell = HistoryCell::Error {
1512 message: "Authentication failed: invalid API key".to_string(),
1513 severity,
1514 };
1515 let lines = cell.lines(80);
1516 assert!(!lines.is_empty(), "{severity:?} must render a line");
1517 let label = &lines[0].spans[0];
1518 (label.content.to_string(), label.style.fg)
1519 };
1520
1521 let (error_label, error_fg) = rank(ErrorSeverity::Error);
1522 let (critical_label, critical_fg) = rank(ErrorSeverity::Critical);
1523 let (warning_label, warning_fg) = rank(ErrorSeverity::Warning);
1524 let (info_label, info_fg) = rank(ErrorSeverity::Info);
1525
1526 assert_eq!(
1527 (critical_label, critical_fg),
1528 (error_label.clone(), error_fg),
1529 "Critical and Error both flip offline mode; they must read identically"
1530 );
1531 assert_ne!(
1532 warning_fg, error_fg,
1533 "a warning that reads as an error is the whole bug this guards"
1534 );
1535 assert_ne!(warning_label, error_label, "and the labels must differ too");
1536 assert_ne!(info_fg, error_fg, "info must not shout");
1537 assert_ne!(info_fg, warning_fg, "info must not read as a warning");
1538 assert_ne!(info_label, warning_label);
1539
1540 // The body inherits the label's rank rather than staying neutral, or the
1541 // colour would carry no information past the first word.
1542 let cell = HistoryCell::Error {
1543 message: "Authentication failed: invalid API key".to_string(),
1544 severity: ErrorSeverity::Error,
1545 };
1546 let body_fg = cell
1547 .lines(80)
1548 .iter()
1549 .flat_map(|line| line.spans.iter())
1550 .find(|span| span.content.contains("Authentication"))
1551 .expect("error body span")
1552 .style
1553 .fg;
1554 assert_eq!(body_fg, error_fg);
1555 }
1556
1557 #[test]
1558 fn error_guidance_keeps_recovery_commands_on_their_own_line() {
1559 let cell = HistoryCell::Error {
1560 message: "DeepSeek API key not found.\nSave it:\n codewhale auth set --provider deepseek"
1561 .to_string(),
1562 severity: crate::error_taxonomy::ErrorSeverity::Error,
1563 };
1564 for width in [80, 140] {
1565 let lines = cell.lines(width);
1566 let command_line = lines
1567 .iter()
1568 .map(|line| {
1569 line.spans
1570 .iter()
1571 .map(|span| span.content.as_ref())
1572 .collect::<String>()
1573 })
1574 .find(|line| line.contains("codewhale auth set --provider deepseek"))
1575 .expect("recovery command stays intact");
1576 assert!(!command_line.contains('\n'), "{command_line:?}");
1577 assert!(!command_line.contains("Save it:"), "{command_line:?}");
1578 }
1579 }
1580
1581 /// A multiline failure can run past the bottom of the terminal while its full
1582 /// text stays in history. The live cell advertises the pager; the pager and the
1583 /// transcript must carry the recovery instruction verbatim and must not
1584 /// recursively advertise themselves.
1585 #[test]
1586 fn an_error_cell_advertises_the_pager_live_and_never_inside_it() {
1587 let recovery = "Refusing insecure base URL 'http://192.168.1.25:8000/v1'.\n\
1588 Loopback hosts (localhost, 127.0.0.1, [::1]) are auto-allowed.\n\
1589 Set CODEWHALE_ALLOW_INSECURE_HTTP=1 only for a trusted LAN host.";
1590 let cell = HistoryCell::Error {
1591 message: recovery.to_string(),
1592 severity: crate::error_taxonomy::ErrorSeverity::Error,
1593 };
1594
1595 let live_text = lines_text(&cell.lines(48));
1596 let transcript_text = lines_text(&cell.transcript_lines(200));
1597 let hint = crate::tui::key_shortcuts::tool_details_shortcut_action_hint("full error");
1598
1599 assert!(live_text.contains(&hint), "{live_text}");
1600 assert!(!transcript_text.contains(&hint), "{transcript_text}");
1601 assert!(
1602 transcript_text.contains("CODEWHALE_ALLOW_INSECURE_HTTP=1"),
1603 "the actionable instruction must survive verbatim: {transcript_text}"
1604 );
1605 assert!(
1606 transcript_text.contains("192.168.1.25"),
1607 "the offending host must survive verbatim: {transcript_text}"
1608 );
1609 }
1610
1611 // ---------------------------------------------------------------------------
1612 // Cards with a content contract
1613 // ---------------------------------------------------------------------------
1614
1615 /// A search receipt has to name where the answer came from and whether the
1616 /// provider it claimed was the provider it used.
1617 #[test]
1618 fn a_web_search_receipt_names_its_source_and_any_degradation() {
1619 let cell = WebSearchCell {
1620 query: "current release".to_string(),
1621 status: ToolStatus::Success,
1622 summary: Some("Found 2 results".to_string()),
1623 source: Some("provider-native/xai/grok-4.5".to_string()),
1624 degraded: Some("provider_native -> duckduckgo".to_string()),
1625 ref_count: 2,
1626 };
1627
1628 let rendered = lines_text(&cell.lines_with_motion(120, true));
1629
1630 for needle in [
1631 "source",
1632 "provider-native/xai/grok-4.5",
1633 "degraded",
1634 "provider_native -> duckduckgo",
1635 "citations",
1636 ] {
1637 assert!(rendered.contains(needle), "missing {needle:?}: {rendered}");
1638 }
1639 }
1640
1641 /// A workflow's transcript is one row per run; live progress is the
1642 /// workbar's. A foreground `run` card that returned its settled record says
1643 /// only the finish — the final state replaces `started` — without repeating the header
1644 /// in a body; the expanded card adds the goal, the child labels, the final
1645 /// result and the error; the status card lists the runs it found.
1646 ///
1647 /// Replaces three tests, and drops assertions of the form
1648 /// `contains('s') || contains('m')` — true of essentially any English string.
1649 #[test]
1650 fn workflow_cards_report_lifecycle_children_phases_and_failures() {
1651 let run_output = serde_json::json!({
1652 "run_id": "workflow_2400c600",
1653 "status": "completed",
1654 "workflow_goal": "audit the FLEET and WORKFLOW docs",
1655 "child_ids": ["a1", "a2", "a3"],
1656 "progress": ["phase: Scan", "log: 3 findings"],
1657 "events": [
1658 {"type": "task_started", "task_id": "a1", "label": "scan-docs",
1659 "workflow_run_id": "workflow_2400c600", "workflow_phase_id": "Scan",
1660 "workflow_task_label": "scan-docs", "workflow_child_index": 0},
1661 {"type": "task_started", "task_id": "a2", "workflow_task_label": "check-fleet",
1662 "workflow_run_id": "workflow_2400c600", "workflow_child_index": 1},
1663 {"type": "task_started", "task_id": "a3", "label": "summarize",
1664 "workflow_run_id": "workflow_2400c600", "workflow_child_index": 2},
1665 ],
1666 "schema_errors": [],
1667 })
1668 .to_string();
1669 let mut run = generic_tool("workflow", ToolStatus::Success);
1670 run.input_summary = Some("action: run".to_string());
1671 run.output = Some(run_output);
1672 let text = lines_text(&run.lines_with_mode(120, true, RenderMode::Live));
1673 assert!(
1674 !text.contains("started"),
1675 "the settled record replaces the start line: {text:?}"
1676 );
1677 assert!(
1678 text.contains("finished") && text.contains("/3 done"),
1679 "the finish line with done/total agents: {text:?}"
1680 );
1681 assert_eq!(text.lines().count(), 1, "one row for the run: {text:?}");
1682 // #6503: a run with no failures does not announce `0 fail`.
1683 assert!(!text.contains("fail"), "no zero failure count: {text:?}");
1684 assert!(
1685 !text.contains("status:"),
1686 "the body must not repeat the header lifecycle: {text:?}"
1687 );
1688
1689 let failed_output = serde_json::json!({
1690 "run_id": "workflow_exp",
1691 "status": "failed",
1692 "workflow_goal": "ship v0.8.68",
1693 "started_at_ms": 1000,
1694 "completed_at_ms": 5000,
1695 "source_path": "workflows/demo.workflow.js",
1696 "error": "phase Verify failed",
1697 "result": {"summary": "2 of 3 children ok"},
1698 "events": [
1699 {"type": "run_started", "at_ms": 1000, "run_id": "workflow_exp",
1700 "workflow_goal": "ship v0.8.68"},
1701 {"type": "phase_started", "at_ms": 1100, "title": "Verify"},
1702 {"type": "task_started", "at_ms": 1200, "task_id": "t1", "label": "run tests",
1703 "workflow_task_label": "run tests", "profile": "implementer"},
1704 {"type": "task_completed", "at_ms": 4000, "task_id": "t1", "status": "failed"},
1705 {"type": "run_completed", "at_ms": 5000, "status": "failed",
1706 "error": "phase Verify failed"}
1707 ]
1708 })
1709 .to_string();
1710 let mut failed = generic_tool("workflow", ToolStatus::Failed);
1711 failed.input_summary = Some("action: run".to_string());
1712 failed.output = Some(failed_output);
1713 failed.spillover_path = Some(PathBuf::from("/tmp/wf-artifact.json"));
1714 let text = lines_text(&failed.lines_with_mode(140, true, RenderMode::Transcript));
1715 for needle in [
1716 "ship v0.8.68",
1717 "Verify",
1718 "run tests",
1719 "2 of 3",
1720 "phase Verify failed",
1721 ] {
1722 assert!(
1723 text.contains(needle),
1724 "the expanded card must carry {needle:?}: {text}"
1725 );
1726 }
1727
1728 let status_output = serde_json::json!({
1729 "action": "status",
1730 "count": 2,
1731 "runs": [
1732 {"run_id": "workflow_aaa", "status": "running", "child_count": 4},
1733 {"run_id": "workflow_bbb", "status": "completed", "child_count": 1},
1734 ],
1735 })
1736 .to_string();
1737 let mut status = generic_tool("workflow", ToolStatus::Success);
1738 status.input_summary = Some("action: status".to_string());
1739 status.output = Some(status_output);
1740 let text = lines_text(&status.lines_with_mode(120, true, RenderMode::Live));
1741 for needle in ["2 run(s)", "workflow_aaa", "running", "workflow_bbb"] {
1742 assert!(
1743 text.contains(needle),
1744 "the status card must list {needle:?}: {text:?}"
1745 );
1746 }
1747 }
1748
1749 /// The founder's transcript (2026-09-28): a refused `start` echoed
1750 /// `action: start` and hid its reason; the settled run's reason was cut
1751 /// mid-word. Each is one row that says why.
1752 #[test]
1753 fn workflow_rows_say_why_without_raw_action_or_mid_word_cuts() {
1754 let mut refused = generic_tool("workflow", ToolStatus::Failed);
1755 refused.input_summary = Some("action: start".to_string());
1756 refused.output = Some(
1757 "Error: Invalid input for tool 'workflow': Workflow leaf 'engine-readiness': \
1758 task(): cwd entries must be bounded repo-relative paths\n\
1759 Tool validation feedback: {\"category\":\"invalid_input\"}"
1760 .to_string(),
1761 );
1762 let text = lines_text(&refused.lines_with_mode(100, true, RenderMode::Live));
1763 assert!(!text.contains("action: start"), "{text}");
1764 assert!(
1765 text.contains("cwd entries must be bounded repo-relative paths"),
1766 "{text}"
1767 );
1768 assert!(!text.contains("Invalid input for tool"), "{text}");
1769
1770 let reason = "[auth] Authorization failed: You have run out of credits or need a Grok \
1771 subscription. Add credits at https://grok.com/?_s=usage.";
1772 let failed = serde_json::json!({
1773 "run_id": "workflow_6409ebe6",
1774 "status": "failed",
1775 "workflow_goal": "Read-only release-readiness audit for Codewhale v0.10.1. Determine blockers.",
1776 "started_at_ms": 1_000,
1777 "completed_at_ms": 1_355,
1778 "error": "no task produced a result: all 2 task(s) failed and 1 fan-out(s) lost every slot (no work survived them); the recorded result reflects no completed work",
1779 "transcript_line": "finished",
1780 "events": [
1781 {"type": "run_started", "at_ms": 1_000, "workflow_goal": "Read-only release-readiness audit for Codewhale v0.10.1. Determine blockers."},
1782 {"type": "task_started", "at_ms": 1_080, "task_id": "a", "workflow_task_label": "engine-readiness"},
1783 {"type": "task_started", "at_ms": 1_117, "task_id": "b", "workflow_task_label": "desktop-readiness"},
1784 {"type": "task_completed", "at_ms": 1_329, "task_id": "a", "status": "failed", "reason": reason},
1785 {"type": "task_completed", "at_ms": 1_338, "task_id": "b", "status": "failed", "reason": reason},
1786 {"type": "run_completed", "at_ms": 1_355, "status": "failed"},
1787 ],
1788 })
1789 .to_string();
1790 let mut finish = generic_tool("workflow", ToolStatus::Failed);
1791 finish.output = Some(failed);
1792 let text = lines_text(&finish.lines_with_mode(60, true, RenderMode::Live));
1793 assert!(!text.contains("started"), "{text}");
1794 assert!(text.contains("0/2 done · 2 failed"), "{text}");
1795 assert!(text.contains("355ms"), "{text}");
1796 // The whole first sentence, wrapped, never cut.
1797 let flat = text
1798 .split_whitespace()
1799 .filter(|word| {
1800 !word
1801 .chars()
1802 .all(|ch| ('\u{2500}'..='\u{259F}').contains(&ch))
1803 })
1804 .collect::<Vec<_>>()
1805 .join(" ");
1806 assert!(
1807 flat.contains(
1808 "Authorization failed: You have run out of credits or need a Grok subscription"
1809 ),
1810 "{text}"
1811 );
1812 assert!(
1813 !flat.contains("grok.com"),
1814 "only the first sentence: {text}"
1815 );
1816 assert!(!text.contains("..."), "{text}");
1817 }
1818
1819 #[test]
1820 fn degraded_workflow_receipt_is_terminal_warning_not_running_or_success() {
1821 let output = serde_json::json!({
1822 "run_id": "workflow_partial",
1823 "status": "degraded",
1824 "workflow_goal": "review the release",
1825 "started_at_ms": 1_000,
1826 "completed_at_ms": 2_000,
1827 "dispatch_failure_count": 1,
1828 "dispatch_failures": [{
1829 "label": "review docs",
1830 "message": "profile unavailable",
1831 "at_ms": 1_500,
1832 }],
1833 })
1834 .to_string();
1835 let mut run = generic_tool("workflow", ToolStatus::Success);
1836 run.output = Some(output);
1837
1838 let lines = run.lines_with_mode(120, false, RenderMode::Live);
1839 let text = lines_text(&lines);
1840 assert!(
1841 text.contains("finished with gaps"),
1842 "warning receipt missing: {text:?}"
1843 );
1844 assert!(
1845 !text.to_lowercase().contains(" done"),
1846 "must not read as success: {text:?}"
1847 );
1848 assert!(
1849 !text.contains(" running"),
1850 "must not read as live: {text:?}"
1851 );
1852
1853 let warning = lines
1854 .iter()
1855 .flat_map(|line| line.spans.iter())
1856 .find(|span| span.content.as_ref() == "finished with gaps")
1857 .expect("terminal warning status span");
1858 assert_eq!(
1859 warning.style.fg,
1860 Some(super::tool_rail_color(ToolStatus::Warning)),
1861 "degraded receipt must use the terminal warning accent"
1862 );
1863 assert!(
1864 lines
1865 .iter()
1866 .flat_map(|line| line.spans.iter())
1867 .all(|span| !span
1868 .content
1869 .chars()
1870 .any(|ch| ('\u{2800}'..='\u{28ff}').contains(&ch))),
1871 "terminal receipt must not retain a spinner: {text:?}"
1872 );
1873 }
1874
1875 /// A checklist update names one item. Showing the rest would make every
1876 /// single-item edit cost the height of the whole list; showing none would make
1877 /// the row unreadable. An id past the end of the list falls back to a
1878 /// placeholder instead of panicking.
1879 #[test]
1880 fn a_checklist_update_shows_only_the_item_that_changed() {
1881 let snapshot = super::ChecklistSnapshot {
1882 items: vec![
1883 super::ChecklistItemSnapshot {
1884 content: "Read the spec".to_string(),
1885 status: "completed".to_string(),
1886 },
1887 super::ChecklistItemSnapshot {
1888 content: "Write the test".to_string(),
1889 status: "in_progress".to_string(),
1890 },
1891 super::ChecklistItemSnapshot {
1892 content: "Land the PR".to_string(),
1893 status: "pending".to_string(),
1894 },
1895 ],
1896 completion_pct: 33,
1897 completed: 1,
1898 total: 3,
1899 };
1900 let lines = super::render_checklist_change_card(
1901 "todo_update",
1902 ToolStatus::Success,
1903 &snapshot,
1904 &super::ChecklistChange {
1905 id: 2,
1906 status: "in_progress".to_string(),
1907 },
1908 80,
1909 true,
1910 );
1911 assert!(lines.len() >= 3, "header, change, summary: {}", lines.len());
1912
1913 let change = line_text(&lines[1]);
1914 for needle in ["#2", "Write the test", "in_progress"] {
1915 assert!(change.contains(needle), "missing {needle:?}: {change:?}");
1916 }
1917 for other in ["Land the PR", "Read the spec"] {
1918 assert!(
1919 !change.contains(other),
1920 "an update must not redraw the whole list: {change:?}"
1921 );
1922 }
1923
1924 let summary = line_text(lines.last().expect("summary row"));
1925 assert!(summary.contains("3 items"), "{summary:?}");
1926 assert!(
1927 summary.contains(&crate::tui::key_shortcuts::tool_details_shortcut_action_hint("list")),
1928 "the full list stays one keypress away: {summary:?}"
1929 );
1930
1931 let single = super::ChecklistSnapshot {
1932 items: vec![super::ChecklistItemSnapshot {
1933 content: "only item".to_string(),
1934 status: "pending".to_string(),
1935 }],
1936 completion_pct: 0,
1937 completed: 0,
1938 total: 1,
1939 };
1940 let lines = super::render_checklist_change_card(
1941 "todo_update",
1942 ToolStatus::Success,
1943 &single,
1944 &super::ChecklistChange {
1945 id: 99,
1946 status: "completed".to_string(),
1947 },
1948 80,
1949 true,
1950 );
1951 let change = line_text(&lines[1]);
1952 assert!(change.contains("#99") && change.contains("(missing title)"));
1953 }
1954
1955 /// The plan card is the only place a plan's supporting artifact is visible.
1956 /// Every populated section has to reach the surface, or the model can record
1957 /// context the user never sees.
1958 #[test]
1959 fn a_plan_card_surfaces_every_populated_artifact_section() {
1960 let cell = PlanUpdateCell {
1961 snapshot: PlanSnapshot {
1962 objective: Some("Make Plan mode reviewable".to_string()),
1963 context_summary: Some("Grounded in issue #2691".to_string()),
1964 sources_used: vec!["gh issue view 2691".to_string()],
1965 critical_files: vec!["crates/tui/src/tools/plan.rs".to_string()],
1966 constraints: vec!["Keep To-do primary".to_string()],
1967 recommended_approach: Some(
1968 "Enrich update_plan without breaking legacy calls".to_string(),
1969 ),
1970 verification_plan: Some("Run focused renderer tests".to_string()),
1971 risks_and_unknowns: Some("Metadata-only plans can disappear".to_string()),
1972 handoff_packet: Some("Next agent should inspect relay output".to_string()),
1973 items: vec![crate::tools::plan::PlanItemArg {
1974 step: "Render artifact sections".to_string(),
1975 status: StepStatus::InProgress,
1976 }],
1977 ..PlanSnapshot::default()
1978 },
1979 status: ToolStatus::Success,
1980 };
1981
1982 let visible = lines_text(&cell.lines_with_motion(120, true));
1983
1984 for needle in [
1985 "objective:",
1986 "Make Plan mode reviewable",
1987 "source:",
1988 "gh issue view 2691",
1989 "file:",
1990 "verify:",
1991 "handoff:",
1992 "Render artifact sections",
1993 ] {
1994 assert!(visible.contains(needle), "missing {needle:?}: {visible}");
1995 }
1996 }
1997
1998 /// A fan-out tool's per-child prompts get one row each so the user can read
1999 /// what each child was asked; the inline `args:` summary that would otherwise
2000 /// say `prompts: <3 items>` is suppressed rather than printed alongside them.
2001 #[test]
2002 fn fan_out_prompts_replace_the_inline_argument_summary() {
2003 let mut cell = generic_tool("read_file", ToolStatus::Running);
2004 cell.input_summary = Some("prompts: <3 items>".to_string());
2005 cell.prompts = Some(vec![
2006 "Summarize the README".to_string(),
2007 "List the public types in client.rs".to_string(),
2008 "Diff this commit against main".to_string(),
2009 ]);
2010 let text = lines_text(&HistoryCell::Tool(ToolCell::Generic(cell)).lines(80));
2011
2012 assert!(text.contains("[0] Summarize the README"));
2013 assert!(text.contains("[1] List the public types in client.rs"));
2014 assert!(text.contains("[2] Diff this commit against main"));
2015 assert!(
2016 !text.contains("args: prompts:"),
2017 "the summary the rows replaced must not also render: {text}"
2018 );
2019
2020 let mut plain = generic_tool("file_search", ToolStatus::Running);
2021 plain.input_summary = Some("query: foo".to_string());
2022 let text = lines_text(&HistoryCell::Tool(ToolCell::Generic(plain)).lines(80));
2023 assert!(
2024 text.contains("query: foo"),
2025 "a non-fan-out tool keeps its argument summary: {text}"
2026 );
2027 }
2028
2029 /// A grouped activity row is metadata, not a tool card: exactly one line, and
2030 /// the synthetic tool name that carries it never reaches the screen.
2031 #[test]
2032 fn an_activity_group_renders_as_a_single_metadata_line() {
2033 let mut cell = generic_tool("activity_group", ToolStatus::Success);
2034 cell.input_summary = Some("Explored 2 files, 1 search".to_string());
2035
2036 let lines = cell.lines_with_mode(120, true, RenderMode::Live);
2037
2038 assert_eq!(lines.len(), 1);
2039 assert_eq!(lines_text(&lines), "Explored 2 files, 1 search ›");
2040 assert!(!lines_text(&lines).contains("activity_group"));
2041 }
2042
2043 // ---------------------------------------------------------------------------
2044 // Replay — wire messages project to the right typed cell
2045 // ---------------------------------------------------------------------------
2046
2047 /// The wire carries a `(reasoning omitted)` placeholder for turns whose
2048 /// reasoning the provider did not return. Replaying it as a reasoning cell
2049 /// would put words in the model's mouth.
2050 #[test]
2051 fn restored_history_drops_the_wire_reasoning_placeholder() {
2052 let message = Message {
2053 role: Role::Assistant,
2054 content: vec![
2055 ContentBlock::Thinking {
2056 thinking: "(reasoning omitted)".to_string(),
2057 signature: None,
2058 state: None,
2059 },
2060 ContentBlock::Thinking {
2061 thinking: "Actual model reasoning".to_string(),
2062 signature: None,
2063 state: None,
2064 },
2065 ],
2066 };
2067
2068 let cells = super::history_cells_from_message(&message);
2069 assert_eq!(cells.len(), 1);
2070 assert!(matches!(
2071 &cells[0],
2072 HistoryCell::Thinking { content, .. } if content == "Actual model reasoning"
2073 ));
2074 }
2075
2076 /// Compaction writes an `<archived_context>` envelope whose attributes are the
2077 /// only record of what was dropped. Attribute parsing must survive spaces and
2078 /// punctuation inside the values, or the summary reads with a mangled range.
2079 #[test]
2080 fn archived_context_metadata_survives_spaces_inside_attribute_values() {
2081 let msg = Message {
2082 role: Role::Assistant,
2083 content: vec![ContentBlock::Text {
2084 text: "<archived_context level=\"1\" range=\"msg 0-128\" tokens=\"2499\" \
2085 density=\"~2,500 tokens\" model=\"deepseek-v4-flash\" \
2086 timestamp=\"2026-04-28T00:00:00Z\">\nSummary body\n</archived_context>"
2087 .to_string(),
2088 cache_control: None,
2089 }],
2090 };
2091
2092 let cells = super::history_cells_from_message(&msg);
2093 assert_eq!(cells.len(), 1);
2094 let HistoryCell::ArchivedContext {
2095 level,
2096 range,
2097 tokens,
2098 density,
2099 model,
2100 timestamp,
2101 summary,
2102 } = &cells[0]
2103 else {
2104 panic!("expected archived context cell, got {:?}", cells[0]);
2105 };
2106
2107 assert_eq!(*level, 1);
2108 assert_eq!(range, "msg 0-128");
2109 assert_eq!(tokens, "2499");
2110 assert_eq!(density, "~2,500 tokens");
2111 assert_eq!(model, "deepseek-v4-flash");
2112 assert_eq!(timestamp, "2026-04-28T00:00:00Z");
2113 assert_eq!(summary, "Summary body");
2114 }
2115
2116 /// Two projections that must not become generic assistant prose: a repair
2117 /// receipt is a system note, and a replayed `update_plan` call rebuilds the
2118 /// typed plan cell with its snapshot intact.
2119 #[test]
2120 fn replay_routes_repair_receipts_and_plan_calls_to_typed_cells() {
2121 let repair = Message {
2122 role: Role::Assistant,
2123 content: vec![ContentBlock::Text {
2124 text: "[tool_history_repair] Repaired 1 crashed tool call(s); quarantined 0 \
2125 duplicate and 0 orphan terminal result(s)."
2126 .to_string(),
2127 cache_control: None,
2128 }],
2129 };
2130 assert!(matches!(
2131 super::history_cells_from_message(&repair).as_slice(),
2132 [HistoryCell::System { content }] if content.starts_with("[tool_history_repair]")
2133 ));
2134
2135 let plan = Message {
2136 role: Role::Assistant,
2137 content: vec![ContentBlock::ToolUse {
2138 execution_id: None,
2139 id: "plan-1".to_string(),
2140 name: "update_plan".to_string(),
2141 input: serde_json::json!({
2142 "objective": "Make Plan mode reviewable",
2143 "sources_used": ["gh issue view 2691"],
2144 "critical_files": ["crates/tui/src/tools/plan.rs"],
2145 "plan": [
2146 { "step": "render replay card", "status": "completed" }
2147 ]
2148 }),
2149 caller: None,
2150 thought_signature: None,
2151 }],
2152 };
2153 let cells = super::history_cells_from_message(&plan);
2154 assert_eq!(cells.len(), 1);
2155 let HistoryCell::Tool(ToolCell::PlanUpdate(cell)) = &cells[0] else {
2156 panic!("expected update_plan replay cell");
2157 };
2158 assert_eq!(cell.status, ToolStatus::Success);
2159 assert_eq!(
2160 cell.snapshot.objective.as_deref(),
2161 Some("Make Plan mode reviewable")
2162 );
2163 assert_eq!(cell.snapshot.sources_used, vec!["gh issue view 2691"]);
2164 assert_eq!(cell.snapshot.items[0].status, StepStatus::Completed);
2165 }
2166
2167 /// The runtime appends a `<turn_meta>` block to the user's message. It is
2168 /// scaffolding and must be hidden — but only when it is the trailing block the
2169 /// runtime appended. A user who types the same tag is quoting, not injecting,
2170 /// and their text must survive verbatim.
2171 #[test]
2172 fn user_history_hides_only_the_trailing_turn_metadata_block() {
2173 let visible = "Explain this literal: <turn_meta>example</turn_meta>";
2174 let turn_meta = concat!(
2175 "<turn_meta>\n",
2176 "Current local date: 2026-07-22\n",
2177 "Input provenance: external_user\n",
2178 "Input authority: external_current_turn\n",
2179 "</turn_meta>",
2180 );
2181 let msg = Message {
2182 role: Role::User,
2183 content: vec![
2184 ContentBlock::Text {
2185 text: visible.to_string(),
2186 cache_control: None,
2187 },
2188 ContentBlock::Text {
2189 text: turn_meta.to_string(),
2190 cache_control: None,
2191 },
2192 ],
2193 };
2194 assert!(matches!(
2195 super::history_cells_from_message(&msg).as_slice(),
2196 [HistoryCell::User { content }] if content == visible
2197 ));
2198
2199 let literal_only = Message {
2200 role: Role::User,
2201 content: vec![ContentBlock::Text {
2202 text: "<turn_meta>user-authored example</turn_meta>".to_string(),
2203 cache_control: None,
2204 }],
2205 };
2206 assert!(matches!(
2207 super::history_cells_from_message(&literal_only).as_slice(),
2208 [HistoryCell::User { content }]
2209 if content == "<turn_meta>user-authored example</turn_meta>"
2210 ));
2211 }
2212
2213 /// "Copy answer" must select the last completed assistant cell and serialize
2214 /// exactly its authored text — no reasoning, no tool bodies, no runtime status,
2215 /// no role marker, and never a half-streamed cell.
2216 #[test]
2217 fn the_answer_projection_copies_authored_text_and_nothing_else() {
2218 use crate::tui::ui_text::history_cell_to_clipboard_text;
2219
2220 let cells = [
2221 HistoryCell::User {
2222 content: "please summarize".to_string(),
2223 },
2224 HistoryCell::Thinking {
2225 content: "private reasoning trace".to_string(),
2226 streaming: false,
2227 duration_secs: Some(1.0),
2228 },
2229 HistoryCell::Tool(ToolCell::Generic({
2230 let mut cell = generic_tool("read_file", ToolStatus::Success);
2231 cell.input_summary = Some("src/lib.rs".to_string());
2232 cell.output = Some("raw tool result body".to_string());
2233 cell
2234 })),
2235 HistoryCell::System {
2236 content: "runtime status note".to_string(),
2237 },
2238 HistoryCell::Assistant {
2239 content: "still streaming partial".to_string(),
2240 streaming: true,
2241 },
2242 HistoryCell::Assistant {
2243 content: "## Final answer\nauthored markdown".to_string(),
2244 streaming: false,
2245 },
2246 ];
2247
2248 let answer = cells
2249 .iter()
2250 .rev()
2251 .find(|cell| cell.is_completed_assistant_answer())
2252 .expect("the completed assistant cell must qualify");
2253 let copied = history_cell_to_clipboard_text(answer, 80);
2254
2255 assert_eq!(copied, "## Final answer\nauthored markdown");
2256 for excluded in [
2257 "please summarize",
2258 "private reasoning trace",
2259 "raw tool result body",
2260 "runtime status note",
2261 "still streaming partial",
2262 ASSISTANT_GLYPH,
2263 ] {
2264 assert!(
2265 !copied.contains(excluded),
2266 "answer copy leaked {excluded:?}"
2267 );
2268 }
2269 }
2270
2271 // ---------------------------------------------------------------------------
2272 // Grouping and small parsers
2273 // ---------------------------------------------------------------------------
2274
2275 fn tool_cell(name: &str, status: ToolStatus) -> HistoryCell {
2276 let mut cell = generic_tool(name, status);
2277 cell.input_summary = Some(format!("args for {name}"));
2278 cell.output = Some(format!("output for {name}"));
2279 HistoryCell::Tool(ToolCell::Generic(cell))
2280 }
2281
2282 /// Collapsing a run of tool cards is only safe when nothing in the run needs
2283 /// the user's eyes: a failure, an in-flight call, or a shell command must all
2284 /// break the group and stay individually visible, and a run shorter than the
2285 /// threshold is not a run at all.
2286 ///
2287 /// Replaces three tests.
2288 #[test]
2289 fn only_contiguous_finished_safe_tool_calls_collapse_into_a_run() {
2290 let history = vec![
2291 HistoryCell::User {
2292 content: "go".to_string(),
2293 },
2294 tool_cell("read_file", ToolStatus::Success),
2295 tool_cell("list_dir", ToolStatus::Success),
2296 tool_cell("web_search", ToolStatus::Success),
2297 HistoryCell::Assistant {
2298 content: "done".to_string(),
2299 streaming: false,
2300 },
2301 ];
2302 let runs = super::detect_tool_runs(&history, 3);
2303 assert_eq!(runs.len(), 1);
2304 assert_eq!((runs[0].start, runs[0].count), (1, 3));
2305 assert_eq!(
2306 runs[0].tool_families,
2307 vec!["read_file", "list_dir", "web_search"]
2308 );
2309 assert_eq!(runs[0].activity.files, 2);
2310 assert_eq!(runs[0].activity.searches, 1);
2311
2312 assert!(
2313 super::detect_tool_runs(
2314 &[
2315 tool_cell("read_file", ToolStatus::Success),
2316 tool_cell("list_dir", ToolStatus::Success),
2317 ],
2318 3
2319 )
2320 .is_empty(),
2321 "a run below the threshold is not collapsed"
2322 );
2323
2324 assert!(
2325 super::detect_tool_runs(
2326 &[
2327 tool_cell("read_file", ToolStatus::Success),
2328 HistoryCell::Assistant {
2329 content: "pause".to_string(),
2330 streaming: false,
2331 },
2332 tool_cell("list_dir", ToolStatus::Success),
2333 tool_cell("web_search", ToolStatus::Success),
2334 ],
2335 3
2336 )
2337 .is_empty(),
2338 "assistant prose breaks the run"
2339 );
2340
2341 // Each of failure, in-flight and shell breaks the group; only the clean
2342 // trailing triple survives.
2343 let mut mixed = Vec::new();
2344 for breaker in [
2345 tool_cell("web_search", ToolStatus::Failed),
2346 tool_cell("web_search", ToolStatus::Running),
2347 HistoryCell::Tool(ToolCell::Exec({
2348 let mut exec = exec_tool("rm -rf target", ToolStatus::Success);
2349 exec.output = Some("ok".to_string());
2350 exec
2351 })),
2352 ] {
2353 mixed.push(tool_cell("read_file", ToolStatus::Success));
2354 mixed.push(tool_cell("list_dir", ToolStatus::Success));
2355 mixed.push(breaker);
2356 }
2357 let tail_start = mixed.len();
2358 mixed.push(tool_cell("read_file", ToolStatus::Success));
2359 mixed.push(tool_cell("list_dir", ToolStatus::Success));
2360 mixed.push(tool_cell("web_search", ToolStatus::Success));
2361
2362 let runs = super::detect_tool_runs(&mixed, 3);
2363 assert_eq!(runs.len(), 1, "only the clean tail collapses: {runs:?}");
2364 assert_eq!((runs[0].start, runs[0].count), (tail_start, 3));
2365 }
2366
2367 /// The one-line summary that replaces a collapsed run is the user's only
2368 /// record of it, so it must name what actually happened — the right verb, the
2369 /// right counts, and only the tool families that belong to each clause.
2370 ///
2371 /// Replaces four tests.
2372 #[test]
2373 fn a_collapsed_run_summary_names_what_actually_happened() {
2374 let run = |families: &[&str], activity: super::ToolRunActivitySummary| super::ToolRun {
2375 start: 4,
2376 count: families.len(),
2377 tool_families: families.iter().map(|f| f.to_string()).collect(),
2378 activity,
2379 };
2380
2381 assert_eq!(
2382 super::tool_run_summary(&run(
2383 &["read_file", "list_dir"],
2384 super::ToolRunActivitySummary {
2385 files: 4,
2386 searches: 1,
2387 ..Default::default()
2388 }
2389 )),
2390 "Explored 4 files, 1 search: read_file, list_dir"
2391 );
2392
2393 assert_eq!(
2394 super::tool_run_summary(&run(
2395 &["read_file", "run_tests", "validate_data"],
2396 super::ToolRunActivitySummary {
2397 files: 2,
2398 commands: 2,
2399 ..Default::default()
2400 }
2401 )),
2402 "Explored 2 files: read_file, ran 2 commands: run_tests, validate_data",
2403 "each clause lists only its own families"
2404 );
2405
2406 assert_eq!(
2407 super::tool_run_summary(&run(
2408 &["session_sync"],
2409 super::ToolRunActivitySummary {
2410 other: 2,
2411 ..Default::default()
2412 }
2413 )),
2414 "Updated metadata",
2415 "a run of tools with no user-facing family falls back to a plain note"
2416 );
2417
2418 // Classification is derived from the real cells, not hand-set counters:
2419 // command tools count as commands, git history tools count as files.
2420 let commands = super::detect_tool_runs(
2421 &[
2422 tool_cell("run_tests", ToolStatus::Success),
2423 tool_cell("run_verifiers", ToolStatus::Success),
2424 tool_cell("validate_data", ToolStatus::Success),
2425 ],
2426 3,
2427 );
2428 assert_eq!(commands[0].activity.commands, 3);
2429 assert_eq!(
2430 super::tool_run_summary(&commands[0]),
2431 "Ran 3 commands: run_tests, run_verifiers, validate_data"
2432 );
2433
2434 let git = super::detect_tool_runs(
2435 &[
2436 tool_cell("git_log", ToolStatus::Success),
2437 tool_cell("git_show", ToolStatus::Success),
2438 tool_cell("git_blame", ToolStatus::Success),
2439 ],
2440 3,
2441 );
2442 assert_eq!(git[0].activity.files, 3);
2443 assert_eq!(
2444 super::tool_run_summary(&git[0]),
2445 "Explored 3 files: git_log, git_show, git_blame"
2446 );
2447 }
2448
2449 /// The small pure helpers behind the cards, as one table each. Every row is a
2450 /// documented input shape or a documented rejection; a helper that guesses on
2451 /// malformed input is worse than one that declines.
2452 ///
2453 /// Replaces ten single-case tests.
2454 #[test]
2455 fn the_card_helpers_accept_their_documented_forms_and_decline_the_rest() {
2456 // Agent ids come out of a JSON body that the renderer must not fully parse.
2457 for (input, expected) in [
2458 (
2459 r#"{"agent_id": "agent-abc12", "nickname": "Beluga"}"#,
2460 Some("agent-abc12"),
2461 ),
2462 (
2463 "{\n \"agent_id\" : \"agent-xyz\",\n \"model\": \"x\"\n}",
2464 Some("agent-xyz"),
2465 ),
2466 (r#"{"nickname": "Orca", "model": "x"}"#, None),
2467 (r#"{"agent_id": "", "model": "x"}"#, None),
2468 ("(not json)", None),
2469 ("", None),
2470 ] {
2471 assert_eq!(
2472 super::extract_agent_id(input),
2473 expected,
2474 "extract_agent_id({input:?})"
2475 );
2476 }
2477
2478 // Checklist update prefixes: both vocabularies, and no guessing.
2479 for (input, expected) in [
2480 (
2481 "Updated todo #3 to in_progress\n{ \"items\": [...] }",
2482 Some(super::ChecklistChange {
2483 id: 3,
2484 status: "in_progress".to_string(),
2485 }),
2486 ),
2487 (
2488 "Updated checklist #7 to completed\n{ \"items\": [] }",
2489 Some(super::ChecklistChange {
2490 id: 7,
2491 status: "completed".to_string(),
2492 }),
2493 ),
2494 ("{ \"items\": [] }", None),
2495 ("Wrote 5 todos\n{}", None),
2496 ("Updated todo #3\n", None),
2497 ("Updated todo #foo to done\n", None),
2498 ] {
2499 assert_eq!(
2500 super::parse_update_prefix(input),
2501 expected,
2502 "parse_update_prefix({input:?})"
2503 );
2504 }
2505
2506 // The elapsed badge appears at three seconds and not before, so quick
2507 // reads and greps do not visually churn.
2508 for secs in [0, 1, 2] {
2509 assert_eq!(running_status_label_with_elapsed(secs), "running");
2510 }
2511 for secs in [3u64, 7, 120] {
2512 assert_eq!(
2513 running_status_label_with_elapsed(secs),
2514 format!("running ({secs}s)")
2515 );
2516 }
2517
2518 // A reasoning summary prefers an explicit Summary block, and otherwise is
2519 // the reasoning itself rather than nothing.
2520 assert_eq!(
2521 extract_reasoning_summary("Thinking...\nSummary: First line\nSecond line\n\nTail")
2522 .expect("summary"),
2523 "First line\nSecond line"
2524 );
2525 assert_eq!(
2526 extract_reasoning_summary("Line one\nLine two").expect("summary"),
2527 "Line one\nLine two"
2528 );
2529 }
2530
2531 /// The card rail is the block's border, so it carries the cell's own state.
2532 /// Property, not token: every status paints a distinct rail, the rail is never
2533 /// left unstyled — an unstyled rail is what shipped before, and it made a
2534 /// failed card look exactly like a finished one from the border in — and it
2535 /// agrees with the glyph beside it on everything that needs attention while
2536 /// receding on a settled card, which is the whole point of the split.
2537 #[test]
2538 fn card_rail_carries_the_cell_status() {
2539 let mut rails = Vec::new();
2540 for status in [
2541 ToolStatus::Running,
2542 ToolStatus::Success,
2543 ToolStatus::Hydrated,
2544 ToolStatus::Warning,
2545 ToolStatus::Failed,
2546 ] {
2547 let cell = exec_tool("cargo test", status);
2548 let lines = cell.render(80, /*low_motion*/ true, RenderMode::Live);
2549 let first = &lines[0];
2550
2551 let rail = first.spans.first().expect("card rail span");
2552 assert!(
2553 matches!(rail.content.as_ref(), "─ " | "╭ "),
2554 "{status:?} lost its card rail: {:?}",
2555 rail.content
2556 );
2557 let rail_color = rail.style.fg.unwrap_or_else(|| {
2558 panic!("{status:?} left the card rail unstyled — the border must follow the state")
2559 });
2560
2561 let glyph_color = first.spans[1]
2562 .style
2563 .fg
2564 .expect("header status glyph must be styled");
2565 assert_eq!(
2566 rail_color, glyph_color,
2567 "{status:?} must read the same on the rail and the glyph"
2568 );
2569
2570 rails.push((status, rail_color));
2571 }
2572
2573 for (i, (status, color)) in rails.iter().enumerate() {
2574 for (other_status, other_color) in &rails[i + 1..] {
2575 assert_ne!(
2576 color, other_color,
2577 "{status:?} and {other_status:?} draw the same rail"
2578 );
2579 }
2580 }
2581 }
2582
2583 /// Finished work shares quiet ink while its glyph shape preserves identity:
2584 /// a passed verify and a completed read must still be distinguishable.
2585 #[test]
2586 fn a_settled_verify_glyph_does_not_read_as_a_settled_read() {
2587 let verify = generic_tool("run_tests", ToolStatus::Success);
2588 let read = generic_tool("read_file", ToolStatus::Success);
2589
2590 let glyph = |cell: &GenericToolCell| {
2591 cell.lines_with_mode_and_locale(
2592 80,
2593 /*low_motion*/ true,
2594 RenderMode::Live,
2595 codewhale_localization::Locale::En,
2596 )[0]
2597 // Rail, shared status mark, then the tool-family identity glyph.
2598 .spans[2]
2599 .clone()
2600 };
2601 let verify = glyph(&verify);
2602 let read = glyph(&read);
2603 assert_ne!(verify.content, read.content, "tool identity stays visible");
2604 assert_eq!(
2605 verify.style.fg, read.style.fg,
2606 "settled work shares quiet ink"
2607 );
2608 }
2609
2610 /// An exploring cell rolls its parallel entries up into one state, and a
2611 /// single failure is never averaged away by its successful siblings. This is
2612 /// the fold that used to exist in three places and could not produce `Failed`
2613 /// in the one that painted the header.
2614 #[test]
2615 fn exploring_cell_status_keeps_the_loudest_terminal_state() {
2616 use super::{ExploringCell, ExploringEntry};
2617
2618 let cell = |statuses: &[ToolStatus]| ExploringCell {
2619 entries: statuses
2620 .iter()
2621 .map(|status| ExploringEntry {
2622 label: "Reading src/foo.rs".to_string(),
2623 status: *status,
2624 })
2625 .collect(),
2626 };
2627
2628 for (statuses, expected) in [
2629 (
2630 vec![ToolStatus::Success, ToolStatus::Success],
2631 ToolStatus::Success,
2632 ),
2633 (
2634 vec![ToolStatus::Success, ToolStatus::Running],
2635 ToolStatus::Running,
2636 ),
2637 (
2638 vec![ToolStatus::Failed, ToolStatus::Running],
2639 ToolStatus::Running,
2640 ),
2641 (
2642 vec![ToolStatus::Success, ToolStatus::Failed],
2643 ToolStatus::Failed,
2644 ),
2645 (
2646 vec![ToolStatus::Warning, ToolStatus::Failed],
2647 ToolStatus::Failed,
2648 ),
2649 (
2650 vec![ToolStatus::Success, ToolStatus::Warning],
2651 ToolStatus::Warning,
2652 ),
2653 (
2654 vec![ToolStatus::Success, ToolStatus::Hydrated],
2655 ToolStatus::Hydrated,
2656 ),
2657 ] {
2658 assert_eq!(
2659 cell(&statuses).status(),
2660 expected,
2661 "{statuses:?} should roll up to {expected:?}"
2662 );
2663 }
2664
2665 // And the cell reports that state through the ToolCell it lives in.
2666 assert_eq!(
2667 ToolCell::Exploring(cell(&[ToolStatus::Success, ToolStatus::Failed])).status(),
2668 Some(ToolStatus::Failed)
2669 );
2670 }
2671
2672 // ---------------------------------------------------------------------------
2673 // Tool-card ink: rail vs glyph
2674 // ---------------------------------------------------------------------------
2675
2676 /// The rail is the card border and follows OMP's rule: every status draws a
2677 /// distinct border. Two statuses sharing a rail is the failure this mapping
2678 /// exists to prevent — it is how `Hydrated` once sat on the running accent and
2679 /// read as live work.
2680 #[test]
2681 fn tool_rail_is_distinct_for_every_status() {
2682 let statuses = [
2683 ToolStatus::Running,
2684 ToolStatus::Success,
2685 ToolStatus::Hydrated,
2686 ToolStatus::Warning,
2687 ToolStatus::Failed,
2688 ];
2689 for (i, status) in statuses.iter().enumerate() {
2690 for other in &statuses[i + 1..] {
2691 assert_ne!(
2692 super::tool_rail_color(*status),
2693 super::tool_rail_color(*other),
2694 "{status:?} and {other:?} draw the same rail"
2695 );
2696 }
2697 }
2698 }
2699
2700 /// Finished rows recede while failures and warnings retain attention ink.
2701 #[test]
2702 fn calm1_settled_headers_are_muted_without_hiding_failures() {
2703 use crate::tui::widgets::tool_card::ToolFamily;
2704 use ratatui::style::Modifier;
2705 for family in [ToolFamily::Read, ToolFamily::Verify] {
2706 for status in [
2707 ToolStatus::Running,
2708 ToolStatus::Success,
2709 ToolStatus::Hydrated,
2710 ToolStatus::Warning,
2711 ToolStatus::Failed,
2712 ] {
2713 assert_eq!(
2714 super::tool_rail_color(status),
2715 super::tool_glyph_color(status, family)
2716 );
2717 }
2718 assert_ne!(
2719 super::tool_glyph_color(ToolStatus::Success, family),
2720 super::tool_glyph_color(ToolStatus::Failed, family)
2721 );
2722 }
2723 assert!(
2724 !super::tool_title_style(ToolStatus::Success)
2725 .add_modifier
2726 .contains(Modifier::BOLD)
2727 );
2728 assert!(
2729 super::tool_title_style(ToolStatus::Failed)
2730 .add_modifier
2731 .contains(Modifier::BOLD)
2732 );
2733 }
2734
2735 /// Issue #5871: `todo_write` replaces the whole list on every call, so a long
2736 /// session stacked full checklist cards that could only be cleared by `/clear`
2737 /// or `/new` — and both of those also drop `api_messages` and the compaction
2738 /// summary. Only the newest snapshot keeps its card; the ones it replaced keep
2739 /// their header (the progress reading) and a details affordance.
2740 #[test]
2741 fn superseded_todo_snapshots_collapse_to_their_header() {
2742 let snapshot = |done: usize| {
2743 let items: Vec<String> = (0..3)
2744 .map(|i| {
2745 let status = if i < done { "completed" } else { "pending" };
2746 format!(r#"{{"content":"step {i}","status":"{status}"}}"#)
2747 })
2748 .collect();
2749 let mut cell = generic_tool("todo_write", ToolStatus::Success);
2750 cell.output = Some(format!(r#"{{"items":[{}]}}"#, items.join(",")));
2751 HistoryCell::Tool(ToolCell::Generic(cell))
2752 };
2753
2754 let mut options = TranscriptRenderOptions {
2755 show_tool_details: true,
2756 ..Default::default()
2757 };
2758 let older = snapshot(1);
2759
2760 let expanded = older.lines_with_options(120, options);
2761 assert!(
2762 expanded.len() > 2,
2763 "the newest snapshot renders its full card: {expanded:?}"
2764 );
2765
2766 options.superseded_work_receipt = true;
2767 let collapsed = older.lines_with_options(120, options);
2768 assert_eq!(
2769 collapsed.len(),
2770 2,
2771 "a replaced snapshot keeps its header plus the details affordance"
2772 );
2773 let header: String = collapsed[0]
2774 .spans
2775 .iter()
2776 .map(|span| span.content.as_ref())
2777 .collect();
2778 assert!(
2779 header.contains("1/3"),
2780 "the collapsed row keeps the progress reading: {header}"
2781 );
2782 }
2783
2784 /// One click is one request to open one file.
2785 ///
2786 /// The old `try_open_file_at_line` looped over every line of the cell and
2787 /// spawned a detached editor per match, so a stack trace or a grep result could
2788 /// launch several at once, all fighting the still-raw-mode TUI for the tty
2789 /// (#6235). The parser now returns the first resolvable reference and nothing
2790 /// else; spawning belongs to `external_editor`.
2791 #[test]
2792 fn first_file_line_reference_returns_one_match_and_resolves_it() {
2793 let dir = tempfile::tempdir().unwrap();
2794 let workspace = dir.path();
2795 std::fs::create_dir_all(workspace.join("src")).unwrap();
2796 std::fs::write(workspace.join("src/first.rs"), "fn a() {}\n").unwrap();
2797 std::fs::write(workspace.join("src/second.rs"), "fn b() {}\n").unwrap();
2798
2799 let text = "note: two frames below\n src/first.rs:12\n src/second.rs:34\n";
2800 let (path, line) = super::first_file_line_reference(text, workspace)
2801 .expect("the first resolvable reference is returned");
2802 assert_eq!(path, workspace.join("src/first.rs"));
2803 assert_eq!(line, 12);
2804 }
2805
2806 /// The forms tools and models print: rustc's `-->` locator with a column,
2807 /// a backticked reference, and one inside a sentence with punctuation.
2808 #[test]
2809 fn file_line_reference_reads_common_forms_on_one_line() {
2810 let dir = tempfile::tempdir().unwrap();
2811 let workspace = dir.path();
2812 std::fs::create_dir_all(workspace.join("src")).unwrap();
2813 std::fs::write(workspace.join("src/a.rs"), "fn a() {}\n").unwrap();
2814 let expected = Some((workspace.join("src/a.rs"), 12));
2815
2816 for line in [
2817 " --> src/a.rs:12:5",
2818 "see `src/a.rs:12` for the loop",
2819 "the bug is in (src/a.rs:12), again.",
2820 "src/a.rs:12: error: mismatched types",
2821 "./src/a.rs:12",
2822 ] {
2823 assert_eq!(
2824 super::file_line_reference(line, workspace),
2825 expected,
2826 "{line:?}"
2827 );
2828 }
2829 assert_eq!(super::file_line_reference("src/a.rs:0", workspace), None);
2830 }
2831
2832 /// Model output is not trusted to name a file: an absolute path outside the
2833 /// workspace and a `../` escape both used to open in `$EDITOR`.
2834 #[test]
2835 fn file_line_reference_refuses_paths_outside_the_workspace() {
2836 let root = tempfile::tempdir().unwrap();
2837 let workspace = root.path().join("ws");
2838 std::fs::create_dir_all(workspace.join("src")).unwrap();
2839 std::fs::write(workspace.join("src/in.rs"), "fn a() {}\n").unwrap();
2840 let outside = root.path().join("outside.rs");
2841 std::fs::write(&outside, "secret\n").unwrap();
2842
2843 let absolute_outside = format!("{}:3", outside.display());
2844 assert_eq!(
2845 super::file_line_reference(&absolute_outside, &workspace),
2846 None
2847 );
2848 assert_eq!(
2849 super::file_line_reference("../outside.rs:3", &workspace),
2850 None
2851 );
2852 assert_eq!(
2853 super::first_file_line_reference(
2854 &format!("{absolute_outside}\n../outside.rs:1\n"),
2855 &workspace
2856 ),
2857 None
2858 );
2859
2860 let absolute_inside = format!("{}:4", workspace.join("src/in.rs").display());
2861 assert_eq!(
2862 super::file_line_reference(&absolute_inside, &workspace),
2863 Some((workspace.join("src/in.rs"), 4)),
2864 "an absolute path inside the workspace still opens"
2865 );
2866 }
2867
2868 /// A link inside the workspace passed the text-only check and `is_file()`
2869 /// followed it, so `vendor -> <outside>` or `notes.md -> <outside file>` in
2870 /// model output offered "Open in editor" on a file outside the workspace.
2871 #[cfg(unix)]
2872 #[test]
2873 fn file_line_reference_refuses_links_out_of_the_workspace() {
2874 let root = tempfile::tempdir().unwrap();
2875 let workspace = root.path().join("ws");
2876 let outside = root.path().join("outside");
2877 std::fs::create_dir_all(workspace.join("src")).unwrap();
2878 std::fs::create_dir_all(&outside).unwrap();
2879 std::fs::write(outside.join("secret.rs"), "secret\n").unwrap();
2880 std::fs::write(workspace.join("src/in.rs"), "fn a() {}\n").unwrap();
2881 std::os::unix::fs::symlink(&outside, workspace.join("vendor")).unwrap();
2882 std::os::unix::fs::symlink(outside.join("secret.rs"), workspace.join("notes.rs")).unwrap();
2883 // A link that stays inside is still a link: refused, not followed.
2884 std::os::unix::fs::symlink(workspace.join("src"), workspace.join("alias")).unwrap();
2885
2886 for line in ["vendor/secret.rs:1", "notes.rs:1", "alias/in.rs:1"] {
2887 assert_eq!(
2888 super::file_line_reference(line, &workspace),
2889 None,
2890 "{line:?}"
2891 );
2892 }
2893 assert_eq!(
2894 super::workspace_file(&workspace, "src"),
2895 None,
2896 "a directory"
2897 );
2898 assert_eq!(
2899 super::file_line_reference("src/in.rs:2", &workspace),
2900 Some((workspace.join("src/in.rs"), 2))
2901 );
2902 }
2903
2904 /// The two escapes the review named: a directory link to `/` and links into
2905 /// an `.ssh` directory. The `.ssh` here is one the test creates outside the
2906 /// workspace with real files in it, so a follow-the-link check would find
2907 /// them and resolve; the test does not depend on the host's own keys.
2908 #[cfg(unix)]
2909 #[test]
2910 fn file_line_reference_refuses_links_to_root_and_ssh() {
2911 let dir = tempfile::tempdir().unwrap();
2912 let workspace = &dir.path().join("ws");
2913 std::fs::create_dir_all(workspace).unwrap();
2914 std::os::unix::fs::symlink("/", workspace.join("rootfs")).unwrap();
2915 let ssh = dir.path().join("home/.ssh");
2916 std::fs::create_dir_all(&ssh).unwrap();
2917 std::fs::write(ssh.join("id_ed25519"), "PRIVATE KEY\n").unwrap();
2918 std::fs::write(ssh.join("config"), "Host *\n").unwrap();
2919 std::os::unix::fs::symlink(ssh.join("id_ed25519"), workspace.join("key.rs")).unwrap();
2920 std::os::unix::fs::symlink(&ssh, workspace.join("ssh")).unwrap();
2921 assert!(workspace.join("key.rs").is_file(), "the file link resolves");
2922 assert!(
2923 workspace.join("ssh/config").is_file(),
2924 "the dir link resolves"
2925 );
2926
2927 for line in [
2928 "rootfs/etc/hosts:1",
2929 "./rootfs/etc/hosts:1",
2930 "key.rs:1",
2931 "ssh/config:1",
2932 "ssh/id_ed25519:1",
2933 ] {
2934 assert_eq!(
2935 super::file_line_reference(line, workspace),
2936 None,
2937 "{line:?}"
2938 );
2939 }
2940 let absolute = workspace.join("rootfs/etc/hosts");
2941 assert_eq!(
2942 super::workspace_file(workspace, absolute.to_str().unwrap()),
2943 None,
2944 "an absolute path through the link"
2945 );
2946 }
2947
2948 #[test]
2949 fn first_file_line_reference_skips_unresolvable_and_malformed_rows() {
2950 let dir = tempfile::tempdir().unwrap();
2951 let workspace = dir.path();
2952 std::fs::create_dir_all(workspace.join("src")).unwrap();
2953 std::fs::write(workspace.join("src/real.rs"), "fn a() {}\n").unwrap();
2954
2955 // A path that does not exist, a bare word, a non-numeric suffix and an
2956 // empty suffix all fall through to the one row that resolves.
2957 let text = concat!(
2958 " src/missing.rs:9\n",
2959 " notafile:12\n",
2960 " src/real.rs:abc\n",
2961 " src/real.rs:\n",
2962 " src/real.rs:7\n",
2963 );
2964 let (path, line) =
2965 super::first_file_line_reference(text, workspace).expect("the only resolvable row wins");
2966 assert_eq!(path, workspace.join("src/real.rs"));
2967 assert_eq!(line, 7);
2968
2969 assert!(
2970 super::first_file_line_reference("no references here\n", workspace).is_none(),
2971 "a cell with nothing to open must report nothing, not a default"
2972 );
2973 }
2974
2975 /// #6601: the project-trust warning is a runtime-owned internal message; the
2976 /// model reads it, the transcript never shows it as the user's words.
2977 #[test]
2978 fn workspace_trust_warning_renders_no_transcript_cell() {
2979 for warning in [Some("untrusted project skills were skipped"), None] {
2980 let message = crate::runtime_handoff::workspace_trust_runtime_message(warning);
2981 assert!(crate::runtime_handoff::is_internal_runtime_handoff(
2982 &message
2983 ));
2984 assert!(
2985 super::history_cells_from_message(&message).is_empty(),
2986 "{warning:?}"
2987 );
2988 }
2989 }
2990
2991 #[test]
2992 fn calm1_failed_tool_preview_matrix_preserves_full_details_and_mcp_identity() {
2993 for width in [40, 60, 80, 140] {
2994 for status in [ToolStatus::Running, ToolStatus::Success, ToolStatus::Failed] {
2995 let output = (0..30)
2996 .map(|i| format!("row {i:02}"))
2997 .collect::<Vec<_>>()
2998 .join("\n");
2999 let mcp = HistoryCell::Tool(ToolCell::Mcp(super::McpToolCell {
3000 tool: "linear_get_issue".into(),
3001 status,
3002 content: Some(output.clone()),
3003 is_image: false,
3004 }));
3005 let mut generic = generic_tool("read_file", status);
3006 generic.output = Some(output);
3007 for (cell, is_mcp) in [
3008 (mcp, true),
3009 (HistoryCell::Tool(ToolCell::Generic(generic)), false),
3010 ] {
3011 let options = TranscriptRenderOptions {
3012 calm_mode: true,
3013 show_tool_details: false,
3014 low_motion: true,
3015 ..Default::default()
3016 };
3017 let live = cell.lines_with_options(width, options);
3018 let text = lines_text(&live);
3019 let full = lines_text(&cell.transcript_lines(width));
3020 assert!(
3021 full.contains("row 15"),
3022 "full details must preserve omitted content"
3023 );
3024 if is_mcp {
3025 assert_eq!(text.matches("linear_get_issue").count(), 1, "{text}");
3026 }
3027 if status == ToolStatus::Failed {
3028 assert!(text.contains("row 00") && text.contains("row 29"), "{text}");
3029 assert!(!text.contains("row 15"), "{text}");
3030 assert!(text.contains("lines omitted"), "{text}");
3031 assert_eq!(
3032 (0..30)
3033 .filter(|i| text.contains(&format!("row {i:02}")))
3034 .count(),
3035 6
3036 );
3037 if is_mcp {
3038 assert!(live.len() <= 8, "{text}");
3039 }
3040 }
3041 }
3042 }
3043 }
3044 }
3045
3046 #[test]
3047 fn calm1_settled_reasoning_is_one_localized_row_and_stays_expandable() {
3048 for (locale, expected) in [
3049 (codewhale_localization::Locale::En, "Thought for 12s"),
3050 (codewhale_localization::Locale::ZhHans, "思考用时 12s"),
3051 ] {
3052 let cell = HistoryCell::Thinking {
3053 content: "private reasoning body\nlast step".into(),
3054 streaming: false,
3055 duration_secs: Some(12.0),
3056 };
3057 let options = TranscriptRenderOptions {
3058 calm_mode: true,
3059 locale,
3060 ..Default::default()
3061 };
3062 let (lines, action) = cell.lines_with_options_folded(80, options, None);
3063 assert_eq!(lines.len(), 1);
3064 assert!(
3065 lines_text(&lines).contains(expected),
3066 "{}",
3067 lines_text(&lines)
3068 );
3069 assert_eq!(action, Some(super::ReasoningAction::Expand));
3070 let expanded = cell
3071 .lines_with_options_folded(80, options, Some(ThinkingFold::Expanded))
3072 .0;
3073 assert!(lines_text(&expanded).contains("private reasoning body"));
3074 }
3075 }
3076
3077 #[test]
3078 fn calm1_calm_and_hidden_details_share_one_card_budget() {
3079 let mut exec = exec_tool("command", ToolStatus::Success);
3080 exec.output = Some(numbered_output(40));
3081 let cell = HistoryCell::Tool(ToolCell::Exec(exec));
3082 let mut rendered = Vec::new();
3083 for (calm_mode, show_tool_details) in [(true, false), (true, true), (false, false)] {
3084 let options = TranscriptRenderOptions {
3085 calm_mode,
3086 show_tool_details,
3087 low_motion: true,
3088 ..Default::default()
3089 };
3090 let lines = cell.lines_with_options(80, options);
3091 assert!(lines.len() <= super::constants::TOOL_SUMMARY_CARD_LINES);
3092 rendered.push(lines_text(&lines));
3093 }
3094 assert!(rendered.windows(2).all(|pair| pair[0] == pair[1]));
3095 }
3096
3096 lines RUST