返回 CodeWhale
markdown_render.rs
根目录 / crates / tui / src / tui / markdown_render.rs
1 //! Markdown rendering for TUI transcript lines.
2 //!
3 //! ## Width-independent parse vs width-dependent render (CX#6)
4 //!
5 //! The previous renderer was a single function `render_markdown(content, width)`
6 //! that scanned the source, classified each line (heading / list / code-fence /
7 //! paragraph / link), and word-wrapped to `Line<'static>` in one pass. That meant
8 //! every terminal resize forced a full re-parse of the source for every visible
9 //! cell — wasted work on the streaming cell whose content is changing anyway.
10 //!
11 //! The codex tui solves this by splitting parse from render. We mirror that:
12 //!
13 //! * [`parse`] turns the markdown source into a [`ParsedMarkdown`] AST: a vector
14 //! of width-independent [`Block`]s. The block kind already records all the
15 //! classification decisions (heading level, list bullet, code block membership)
16 //! that don't depend on width.
17 //! * [`render_parsed`] takes a `ParsedMarkdown` plus a width and a base style and
18 //! produces `Vec<Line<'static>>`. It only does word-wrap and span styling.
19 //!
20 //! [`render_markdown`] is kept as a thin convenience that does both — useful for
21 //! callers (Thinking body, message body) that don't want to manage the cache.
22 //!
23 //! The transcript cache layer (see `tui/transcript.rs`) caches the parsed AST per
24 //! cell and re-runs only the render step on width changes. That makes resize a
25 //! re-flow operation rather than a re-parse + re-flow operation.
26
27 #[cfg(test)]
28 use std::cell::Cell;
29 use std::cell::RefCell;
30 use std::sync::OnceLock;
31
32 use ratatui::style::{Color, Modifier, Style};
33 use ratatui::text::{Line, Span};
34 use syntect::easy::HighlightLines;
35 use syntect::highlighting::{FontStyle, HighlightState, Theme, ThemeSet};
36 use syntect::parsing::{ParseState as SyntectParseState, SyntaxSet};
37 use unicode_segmentation::UnicodeSegmentation;
38 use unicode_width::{UnicodeWidthChar, UnicodeWidthStr};
39
40 use crate::tui::osc8;
41 use crate::tui::ui_text::CopyLineSeparator;
42 use codewhale_palette as palette;
43
44 // Thread-local counter incremented every time `parse` runs. Used by tests to
45 // prove that width-only changes hit the cached-AST path and skip parsing.
46 // Thread-local (not global atomic) so concurrent tests calling `parse()` can't
47 // pollute each other's counters.
48 #[cfg(test)]
49 thread_local! {
50 static PARSE_INVOCATIONS: Cell<u64> = const { Cell::new(0) };
51 }
52
53 #[cfg(test)]
54 #[must_use]
55 pub fn parse_invocation_count() -> u64 {
56 PARSE_INVOCATIONS.with(|c| c.get())
57 }
58
59 #[cfg(test)]
60 pub fn reset_parse_invocation_count() {
61 PARSE_INVOCATIONS.with(|c| c.set(0));
62 }
63
64 /// One classified line of markdown source, width-independent.
65 ///
66 /// All decisions that depend only on the source text (heading level, bullet
67 /// kind, whether we're inside a fenced code block, paragraph text) are made at
68 /// parse time. Width-dependent layout (word-wrap, prefix indent) is deferred to
69 /// the render step.
70 #[derive(Debug, Clone, PartialEq, Eq)]
71 pub enum Block {
72 /// `# heading text`. Includes the heading level (1..6).
73 Heading { level: usize, text: String },
74 /// A horizontal rule emitted under a level-1 heading.
75 HeadingRule,
76 /// A standalone `---` / `***` / `___` horizontal rule.
77 HorizontalRule,
78 /// A bullet (`-`/`*`) or ordered (`1.`) list item with its prefix and body.
79 ListItem { bullet: String, text: String },
80 /// A `>` quote line with its nesting depth (1 = single `>`). The text has
81 /// the quote markers stripped; depth is capped at [`MAX_QUOTE_DEPTH`].
82 Quote { depth: usize, text: String },
83 /// A line inside a fenced code block. Fences themselves are dropped, but
84 /// their language token and block identity stay available to syntect.
85 Code {
86 line: String,
87 language: Option<String>,
88 block_id: usize,
89 },
90 /// A table row: cells split on `|`.
91 TableRow(Vec<String>),
92 /// A table separator row (`|---|---|`). Kept so the renderer can draw
93 /// horizontal rules at the correct positions.
94 TableSeparator,
95 /// A non-empty paragraph line that may contain inline links.
96 Paragraph { text: String },
97 /// An empty source line, preserved so paragraph spacing survives.
98 Blank,
99 }
100
101 /// Width-independent parsed-markdown AST for one cell's source.
102 ///
103 /// Wrapped in `Arc` at the cache layer so the cache can hand the same AST to
104 /// many render calls without copying.
105 #[derive(Debug, Clone, PartialEq, Eq)]
106 pub struct ParsedMarkdown {
107 blocks: Vec<Block>,
108 }
109
110 /// Width-dependent rendered line plus the source block kind that produced it.
111 ///
112 /// Most callers only need styled terminal lines, but transcript rendering also
113 /// needs to avoid adding its conversational continuation rail in front of code
114 /// blocks. Keeping this metadata here avoids guessing from styled spans.
115 #[derive(Debug, Clone)]
116 pub struct RenderedMarkdownLine {
117 pub line: Line<'static>,
118 /// Hyperlinks aligned to display columns in `line`. Targets stay
119 /// out-of-band; `Span::content` always contains visible text only.
120 pub links: Vec<osc8::LineLink>,
121 pub is_code: bool,
122 pub copy_prefix_width: usize,
123 pub copy_separator_after: CopyLineSeparator,
124 }
125
126 static SYNTAX_SET: OnceLock<SyntaxSet> = OnceLock::new();
127 static THEME_SET: OnceLock<ThemeSet> = OnceLock::new();
128 static COLOR_DEPTH: OnceLock<palette::ColorDepth> = OnceLock::new();
129 static PALETTE_MODE: OnceLock<palette::PaletteMode> = OnceLock::new();
130
131 fn syntax_set() -> &'static SyntaxSet {
132 SYNTAX_SET.get_or_init(SyntaxSet::load_defaults_newlines)
133 }
134
135 fn theme_set() -> &'static ThemeSet {
136 THEME_SET.get_or_init(ThemeSet::load_defaults)
137 }
138
139 /// Load the syntect syntax and theme sets ahead of the first fenced code
140 /// block, so that render does not pay the one-time deserialization cost.
141 /// Idempotent; intended to run once on a background thread at TUI boot.
142 pub(crate) fn prewarm_syntax_highlighting() {
143 let _ = syntax_set();
144 let _ = theme_set();
145 }
146
147 fn syntax_color_depth() -> palette::ColorDepth {
148 *COLOR_DEPTH.get_or_init(palette::ColorDepth::detect)
149 }
150
151 pub(crate) fn detected_palette_mode() -> palette::PaletteMode {
152 *PALETTE_MODE.get_or_init(palette::PaletteMode::detect)
153 }
154
155 /// Parse markdown source into a width-independent block AST.
156 ///
157 /// This is a small line-oriented parser tuned for the patterns we render:
158 /// fenced code blocks, ATX headings, dash/star/numbered list items, and plain
159 /// paragraphs with optional links. It does not attempt to handle every CommonMark
160 /// edge case — that's intentional. The renderer will treat anything we don't
161 /// classify as `Block::Paragraph`.
162 #[must_use]
163 pub fn parse(content: &str) -> ParsedMarkdown {
164 #[cfg(test)]
165 PARSE_INVOCATIONS.with(|c| c.set(c.get() + 1));
166
167 STREAM_PARSE_MEMO.with(|memo| {
168 let mut memo = memo.borrow_mut();
169 // Reuse the committed prefix when this call continues the same source
170 // (the streaming case). Anything else — a different cell, a shrunk
171 // buffer, an edit to earlier bytes — fails the check and starts clean.
172 let state = memo.get_or_insert_with(ParseState::default);
173 if !state.can_resume_from(content) {
174 *state = ParseState::default();
175 }
176 state.commit_complete_lines(content);
177 let parsed = state.snapshot(content);
178 // Don't hold a whole large message alive between unrelated renders.
179 if state.consumed > MAX_MEMOIZED_PREFIX_BYTES {
180 *state = ParseState::default();
181 }
182 parsed
183 })
184 }
185
186 /// Upper bound on the source we keep memoized between `parse` calls. Streaming
187 /// messages are the reason this exists; past this size the memory cost of
188 /// holding the prefix outweighs the re-parse it saves.
189 const MAX_MEMOIZED_PREFIX_BYTES: usize = 1024 * 1024;
190
191 thread_local! {
192 /// Single-entry resume memo for the streaming re-parse (#3897).
193 ///
194 /// Deliberately one entry and thread-local: the hot path is one cell
195 /// growing chunk by chunk on the render thread. A miss costs exactly what
196 /// the old code always paid, so this can only make things faster or
197 /// identical — never wrong, because [`ParseState::can_resume_from`]
198 /// verifies the prefix byte-for-byte before reusing anything.
199 static STREAM_PARSE_MEMO: RefCell<Option<ParseState>> = const { RefCell::new(None) };
200 }
201
202 /// Resumable parser state.
203 ///
204 /// The parser is strictly line-oriented: each source line maps to blocks using
205 /// only a three-field carry (`open_fence_len`, `code_language`,
206 /// `code_block_id`). That is what makes resuming *exact* rather than
207 /// approximate — appending text can never change how an earlier complete line
208 /// parsed, so committed blocks never need revisiting.
209 ///
210 /// Streaming is the case that matters (#3897): the renderer re-parses the whole
211 /// growing message on every chunk, which is quadratic over message length.
212 #[derive(Debug, Clone, Default)]
213 pub struct ParseState {
214 blocks: Vec<Block>,
215 /// The exact source bytes already folded into `blocks`. Kept verbatim so
216 /// resumption is *verified* against the new content rather than assumed —
217 /// a caller that hands over unrelated text gets a full re-parse, not
218 /// silently wrong output.
219 prefix: String,
220 /// FNV-1a digest of every byte committed since the last reset. The live
221 /// incremental cache drops `prefix` to bound memory, so the
222 /// verified-append resume path compares this digest against the incoming
223 /// content instead of retaining a second copy of the source.
224 committed_digest: u64,
225 /// Length of `prefix`. Always ends just past a newline, so only whole
226 /// lines are ever committed.
227 consumed: usize,
228 /// Length of the opening code fence in backticks while inside a fenced
229 /// code block (`None` outside). A closing fence must be at least this
230 /// long per CommonMark; shorter backtick lines are code content.
231 open_fence_len: Option<usize>,
232 code_language: Option<String>,
233 code_block_id: usize,
234 }
235
236 impl ParseState {
237 /// Fold every *complete* line after `consumed` into `blocks`.
238 ///
239 /// The trailing partial line is deliberately left uncommitted: streaming
240 /// can still extend it, and committing it early would be the one way this
241 /// could diverge from a full re-parse.
242 fn commit_complete_lines(&mut self, content: &str) {
243 let Some(rest) = content.get(self.consumed..) else {
244 return;
245 };
246 let Some(last_newline) = rest.rfind('\n') else {
247 return;
248 };
249 let complete = &rest[..=last_newline];
250 for raw_line in complete.lines() {
251 push_parsed_line(
252 raw_line,
253 &mut self.blocks,
254 &mut self.open_fence_len,
255 &mut self.code_language,
256 &mut self.code_block_id,
257 );
258 }
259 let mut digest = if self.consumed == 0 {
260 committed_prefix_digest(&[])
261 } else {
262 self.committed_digest
263 };
264 for &byte in complete.as_bytes() {
265 digest ^= u64::from(byte);
266 digest = digest.wrapping_mul(FNV_1A_PRIME);
267 }
268 self.committed_digest = digest;
269 self.prefix.push_str(complete);
270 self.consumed += complete.len();
271 }
272
273 /// The full AST: committed blocks plus the trailing partial line, parsed
274 /// against a throwaway copy of the carry so `self` stays resumable.
275 fn snapshot(&self, content: &str) -> ParsedMarkdown {
276 let tail = content.get(self.consumed..).unwrap_or_default();
277 if tail.is_empty() {
278 return ParsedMarkdown {
279 blocks: self.blocks.clone(),
280 };
281 }
282 let mut blocks = self.blocks.clone();
283 let mut open_fence_len = self.open_fence_len;
284 let mut code_language = self.code_language.clone();
285 let mut code_block_id = self.code_block_id;
286 for raw_line in tail.lines() {
287 push_parsed_line(
288 raw_line,
289 &mut blocks,
290 &mut open_fence_len,
291 &mut code_language,
292 &mut code_block_id,
293 );
294 }
295 ParsedMarkdown { blocks }
296 }
297
298 /// True when `content` still starts with everything already committed.
299 ///
300 /// Streaming only ever appends, so this is the common case. An edit that
301 /// rewrites earlier bytes (a re-render of a different cell, a retry) fails
302 /// here and the caller falls back to a full parse — correctness never
303 /// depends on the caller guessing right.
304 fn can_resume_from(&self, content: &str) -> bool {
305 content.len() >= self.consumed
306 && content.is_char_boundary(self.consumed)
307 && self.committed_prefix_matches(content)
308 }
309
310 /// Resume after the caller has proved that the raw source mutation was an
311 /// append. The live transcript obtains that proof at the `push_str` seam.
312 ///
313 /// The receipt covers the *raw* stream, but this cache consumes the
314 /// latex-rendered projection of it, and a math block that closes late
315 /// rewrites already-committed bytes: an open `\[` is committed as literal
316 /// text and only becomes its rendered form once the closing `\]` arrives
317 /// (#6196). A length check cannot see that rewrite, so the committed
318 /// prefix is verified by digest. FNV-1a over hot cache lines costs orders
319 /// of magnitude less per beat than the render it guards, while retaining
320 /// the prefix itself would pin the whole message in memory.
321 fn can_resume_verified_append(&self, content: &str) -> bool {
322 content.len() >= self.consumed
323 && content.is_char_boundary(self.consumed)
324 && self.committed_digest
325 == committed_prefix_digest(&content.as_bytes()[..self.consumed])
326 }
327
328 fn committed_prefix_matches(&self, content: &str) -> bool {
329 self.prefix == content[..self.consumed]
330 }
331 }
332
333 /// FNV-1a constants. Chosen for speed and zero dependencies: the
334 /// verified-append resume check hashes the whole committed prefix on every
335 /// streaming beat, so the guard must stay far below the render cost it
336 /// protects.
337 const FNV_1A_PRIME: u64 = 0x0000_0100_0000_01b3;
338
339 fn committed_prefix_digest(bytes: &[u8]) -> u64 {
340 const FNV_1A_OFFSET_BASIS: u64 = 0xcbf2_9ce4_8422_2325;
341 let mut digest = FNV_1A_OFFSET_BASIS;
342 for &byte in bytes {
343 digest ^= u64::from(byte);
344 digest = digest.wrapping_mul(FNV_1A_PRIME);
345 }
346 digest
347 }
348
349 /// Deterministic work receipts for the live incremental renderer.
350 ///
351 /// These count source lines classified and stable/tail blocks rendered. They
352 /// deliberately do not use wall-clock time, allocator counters, or sampling,
353 /// so regression tests are stable on every machine.
354 #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
355 pub(crate) struct MarkdownRenderWork {
356 pub classified_lines: u64,
357 pub stable_blocks_rendered: u64,
358 pub tail_blocks_rendered: u64,
359 pub invalidations: u64,
360 }
361
362 #[derive(Debug, Clone, Copy, PartialEq, Eq)]
363 struct IncrementalRenderKey {
364 width: u16,
365 base_style: Style,
366 palette_mode: palette::PaletteMode,
367 }
368
369 #[derive(Debug, Clone)]
370 struct IncrementalCodeHighlighter {
371 block_id: usize,
372 language: Option<String>,
373 state: Option<(HighlightState, SyntectParseState)>,
374 }
375
376 /// Persistent state for the one changing Markdown cell in the transcript.
377 ///
378 /// Stable rendered lines live in the owning `CachedCell`; this object retains
379 /// only parser/highlighter carry plus the line index at which its replaceable
380 /// tail begins. That lets each append truncate and replace the old tail without
381 /// cloning or re-rendering the committed prefix.
382 #[derive(Debug, Default)]
383 pub(crate) struct IncrementalMarkdownRenderCache {
384 parser: ParseState,
385 key: Option<IncrementalRenderKey>,
386 source_len: usize,
387 stable_rendered_line_count: usize,
388 code_highlighter: Option<IncrementalCodeHighlighter>,
389 work: MarkdownRenderWork,
390 }
391
392 pub(crate) struct IncrementalMarkdownRenderDelta {
393 pub replace_from: usize,
394 pub lines: Vec<RenderedMarkdownLine>,
395 }
396
397 impl IncrementalMarkdownRenderCache {
398 #[cfg(test)]
399 #[must_use]
400 pub(crate) fn work(&self) -> MarkdownRenderWork {
401 self.work
402 }
403
404 #[cfg(test)]
405 #[must_use]
406 pub(crate) fn retained_source_bytes(&self) -> usize {
407 self.parser.prefix.len()
408 }
409
410 /// Update from a source mutation whose append-only provenance was recorded
411 /// by the live event loop. `verified_append` must be false for edits,
412 /// replacement text, cell reuse, or any unknown mutation.
413 pub(crate) fn update(
414 &mut self,
415 content: &str,
416 width: u16,
417 base_style: Style,
418 palette_mode: palette::PaletteMode,
419 verified_append: bool,
420 ) -> IncrementalMarkdownRenderDelta {
421 let key = IncrementalRenderKey {
422 width: width.max(1),
423 base_style,
424 palette_mode,
425 };
426
427 let can_resume = self.key == Some(key)
428 && verified_append
429 && content.len() >= self.source_len
430 && self.parser.can_resume_verified_append(content);
431 let replace_from = if can_resume {
432 self.stable_rendered_line_count
433 } else {
434 self.reset_for_invalidation();
435 self.work.invalidations = self.work.invalidations.saturating_add(1);
436 0
437 };
438 self.key = Some(key);
439
440 let before_consumed = self.parser.consumed;
441 self.parser.commit_complete_lines(content);
442 self.work.classified_lines = self.work.classified_lines.saturating_add(
443 content[before_consumed..self.parser.consumed]
444 .lines()
445 .count() as u64,
446 );
447 self.source_len = content.len();
448 // Append provenance is carried by the event-loop receipt, so the live
449 // cache does not need a second copy of already committed source. The
450 // absolute byte offset and parser carry are enough to resume.
451 self.parser.prefix.clear();
452
453 let stable_end = stable_block_prefix_len(&self.parser.blocks);
454 let mut lines = self.render_stable_prefix(stable_end, key);
455 self.stable_rendered_line_count =
456 self.stable_rendered_line_count.saturating_add(lines.len());
457
458 let mut tail_blocks = self.parser.blocks.clone();
459 tail_blocks.extend(self.parser.snapshot_tail(content));
460 if !tail_blocks.is_empty() {
461 self.work.tail_blocks_rendered = self
462 .work
463 .tail_blocks_rendered
464 .saturating_add(tail_blocks.len() as u64);
465 let mut tail_highlighter = self.code_highlighter.clone();
466 lines.extend(render_incremental_blocks(
467 &tail_blocks,
468 key,
469 &mut tail_highlighter,
470 ));
471 }
472
473 if lines.is_empty() && self.stable_rendered_line_count == 0 {
474 lines.push(empty_rendered_markdown_line());
475 }
476
477 IncrementalMarkdownRenderDelta {
478 replace_from,
479 lines,
480 }
481 }
482
483 fn render_stable_prefix(
484 &mut self,
485 end: usize,
486 key: IncrementalRenderKey,
487 ) -> Vec<RenderedMarkdownLine> {
488 if end == 0 {
489 return Vec::new();
490 }
491 self.work.stable_blocks_rendered =
492 self.work.stable_blocks_rendered.saturating_add(end as u64);
493 let lines =
494 render_incremental_blocks(&self.parser.blocks[..end], key, &mut self.code_highlighter);
495 self.parser.blocks.drain(..end);
496 lines
497 }
498
499 fn reset_for_invalidation(&mut self) {
500 self.parser = ParseState::default();
501 self.key = None;
502 self.source_len = 0;
503 self.stable_rendered_line_count = 0;
504 self.code_highlighter = None;
505 }
506 }
507
508 impl ParseState {
509 fn snapshot_tail(&self, content: &str) -> Vec<Block> {
510 let tail = content.get(self.consumed..).unwrap_or_default();
511 let mut blocks = Vec::new();
512 let mut open_fence_len = self.open_fence_len;
513 let mut code_language = self.code_language.clone();
514 let mut code_block_id = self.code_block_id;
515 for raw_line in tail.lines() {
516 push_parsed_line(
517 raw_line,
518 &mut blocks,
519 &mut open_fence_len,
520 &mut code_language,
521 &mut code_block_id,
522 );
523 }
524 blocks
525 }
526 }
527
528 fn stable_block_prefix_len(blocks: &[Block]) -> usize {
529 let Some(last_non_table) = blocks
530 .iter()
531 .rposition(|block| !matches!(block, Block::TableRow(_) | Block::TableSeparator))
532 else {
533 return 0;
534 };
535 if last_non_table + 1 == blocks.len() {
536 blocks.len()
537 } else {
538 last_non_table + 1
539 }
540 }
541
542 /// Classify one source line into blocks, advancing the fenced-code carry.
543 ///
544 /// Extracted from the original loop body unchanged so the batch and streaming
545 /// paths cannot drift: both call exactly this.
546 fn push_parsed_line(
547 raw_line: &str,
548 blocks: &mut Vec<Block>,
549 open_fence_len: &mut Option<usize>,
550 code_language: &mut Option<String>,
551 code_block_id: &mut usize,
552 ) {
553 let trimmed = raw_line.trim_start();
554 let fence_len = trimmed.chars().take_while(|c| *c == '`').count();
555 if fence_len >= 3 {
556 match *open_fence_len {
557 // Inside a code block: a fence at least as long as the opener
558 // closes it; a shorter backtick line is code content per
559 // CommonMark and must not flip the state or escape the block.
560 Some(open) if fence_len >= open && trimmed[fence_len..].trim().is_empty() => {
561 *open_fence_len = None;
562 *code_language = None;
563 }
564 Some(_) => {
565 blocks.push(Block::Code {
566 line: raw_line.to_string(),
567 language: code_language.clone(),
568 block_id: *code_block_id,
569 });
570 }
571 None => {
572 *open_fence_len = Some(fence_len);
573 *code_block_id = code_block_id.saturating_add(1);
574 *code_language = normalized_fence_language(&trimmed[fence_len..]);
575 }
576 }
577 return;
578 }
579
580 if open_fence_len.is_some() {
581 blocks.push(Block::Code {
582 line: raw_line.to_string(),
583 language: code_language.clone(),
584 block_id: *code_block_id,
585 });
586 return;
587 }
588
589 if let Some((depth, text)) = parse_blockquote(trimmed) {
590 blocks.push(Block::Quote {
591 depth,
592 text: text.to_string(),
593 });
594 return;
595 }
596
597 if let Some((level, text)) = parse_heading(trimmed) {
598 blocks.push(Block::Heading {
599 level,
600 text: text.to_string(),
601 });
602 if level == 1 {
603 blocks.push(Block::HeadingRule);
604 }
605 return;
606 }
607
608 if let Some((bullet, text)) = parse_list_item(trimmed) {
609 blocks.push(Block::ListItem {
610 bullet,
611 text: text.to_string(),
612 });
613 return;
614 }
615
616 if is_horizontal_rule(trimmed) {
617 blocks.push(Block::HorizontalRule);
618 return;
619 }
620
621 match parse_table_row(trimmed) {
622 Some(cells) => {
623 blocks.push(Block::TableRow(cells));
624 return;
625 }
626 None if trimmed.starts_with('|') => {
627 blocks.push(Block::TableSeparator);
628 return;
629 }
630 None => {}
631 }
632
633 if trimmed.is_empty() {
634 // Whitespace-only lines are blank paragraphs.
635 blocks.push(Block::Blank);
636 return;
637 }
638
639 blocks.push(Block::Paragraph {
640 text: raw_line.to_string(),
641 });
642 }
643
644 /// Render a parsed-markdown AST at the given terminal width.
645 ///
646 /// This is the width-dependent half: word-wrapping, link styling, code-block
647 /// formatting. The AST is owned by the caller (typically the transcript cache),
648 /// so width-only changes can call `render_parsed` again with the same AST and
649 /// skip the parse step entirely.
650 #[must_use]
651 pub fn render_parsed(parsed: &ParsedMarkdown, width: u16, base_style: Style) -> Vec<Line<'static>> {
652 render_parsed_tagged_with_palette(parsed, width, base_style, detected_palette_mode())
653 .into_iter()
654 .map(|line| line.line)
655 .collect()
656 }
657
658 /// Render a parsed-markdown AST and preserve per-line source metadata.
659 #[cfg(test)]
660 #[must_use]
661 pub fn render_parsed_tagged(
662 parsed: &ParsedMarkdown,
663 width: u16,
664 base_style: Style,
665 ) -> Vec<RenderedMarkdownLine> {
666 render_parsed_tagged_with_palette(parsed, width, base_style, detected_palette_mode())
667 }
668
669 /// Render parsed markdown using the caller's resolved UI palette mode.
670 ///
671 /// The live transcript uses this entry point so an explicit theme selection
672 /// wins over terminal/OS auto-detection and participates in cache invalidation.
673 #[must_use]
674 pub(crate) fn render_parsed_tagged_with_palette(
675 parsed: &ParsedMarkdown,
676 width: u16,
677 base_style: Style,
678 palette_mode: palette::PaletteMode,
679 ) -> Vec<RenderedMarkdownLine> {
680 let width = width.max(1) as usize;
681 let mut out: Vec<RenderedMarkdownLine> = Vec::with_capacity(parsed.blocks.len());
682
683 let mut i = 0;
684 while i < parsed.blocks.len() {
685 if matches!(
686 &parsed.blocks[i],
687 Block::TableRow(_) | Block::TableSeparator
688 ) {
689 let start = i;
690 while i < parsed.blocks.len()
691 && matches!(
692 &parsed.blocks[i],
693 Block::TableRow(_) | Block::TableSeparator
694 )
695 {
696 i += 1;
697 }
698 out.extend(
699 render_table_group(&parsed.blocks[start..i], width, base_style)
700 .into_iter()
701 .map(|line| RenderedMarkdownLine {
702 line,
703 links: Vec::new(),
704 is_code: false,
705 copy_prefix_width: 0,
706 copy_separator_after: CopyLineSeparator::Newline,
707 }),
708 );
709 continue;
710 }
711
712 if let Block::Code {
713 language, block_id, ..
714 } = &parsed.blocks[i]
715 {
716 let start = i;
717 while i < parsed.blocks.len()
718 && matches!(
719 &parsed.blocks[i],
720 Block::Code {
721 block_id: candidate,
722 ..
723 } if candidate == block_id
724 )
725 {
726 i += 1;
727 }
728 let source_lines = parsed.blocks[start..i]
729 .iter()
730 .filter_map(|block| match block {
731 Block::Code { line, .. } => Some(line.as_str()),
732 _ => None,
733 })
734 .collect::<Vec<_>>();
735 let highlighted =
736 highlight_code_block(language.as_deref(), &source_lines, base_style, palette_mode);
737 for spans in highlighted {
738 out.extend(render_wrapped_code_spans_tagged(spans, width));
739 }
740 continue;
741 }
742
743 match &parsed.blocks[i] {
744 Block::Heading { text, .. } => {
745 let style = Style::default()
746 .fg(palette::WHALE_ACTION)
747 .add_modifier(Modifier::BOLD);
748 out.extend(render_wrapped_line_tagged(text, width, style, false, false));
749 }
750 Block::HeadingRule => {
751 out.push(RenderedMarkdownLine {
752 line: Line::from(Span::styled(
753 "─".repeat(width.min(40)),
754 Style::default().fg(palette::TEXT_DIM),
755 )),
756 links: Vec::new(),
757 is_code: false,
758 copy_prefix_width: 0,
759 copy_separator_after: CopyLineSeparator::Newline,
760 });
761 }
762 Block::HorizontalRule => {
763 out.push(RenderedMarkdownLine {
764 line: Line::from(Span::styled(
765 "─".repeat(width.min(60)),
766 Style::default().fg(palette::TEXT_DIM),
767 )),
768 links: Vec::new(),
769 is_code: false,
770 copy_prefix_width: 0,
771 copy_separator_after: CopyLineSeparator::Newline,
772 });
773 }
774 Block::ListItem { bullet, text } => {
775 let bullet_style = Style::default().fg(palette::WHALE_ACTION);
776 out.extend(render_list_line_tagged(
777 bullet,
778 text,
779 width,
780 bullet_style,
781 base_style,
782 ));
783 }
784 Block::Code { .. } => unreachable!(),
785 Block::Quote { depth, text } => {
786 let rail_style = Style::default().fg(palette::WHALE_ACTION);
787 let text_style = Style::default().fg(palette::TEXT_DIM);
788 out.extend(render_quote_line_tagged(
789 text, *depth, width, rail_style, text_style,
790 ));
791 }
792 Block::Paragraph { text } => {
793 let link_style = Style::default()
794 .fg(palette::WHALE_ACTION)
795 .add_modifier(Modifier::UNDERLINED);
796 out.extend(render_line_with_links_tagged(
797 text, width, base_style, link_style,
798 ));
799 }
800 Block::Blank => {
801 out.push(RenderedMarkdownLine {
802 line: Line::from(""),
803 links: Vec::new(),
804 is_code: false,
805 copy_prefix_width: 0,
806 copy_separator_after: CopyLineSeparator::Newline,
807 });
808 }
809 Block::TableRow(_) | Block::TableSeparator => unreachable!(),
810 }
811 i += 1;
812 }
813
814 if out.is_empty() {
815 out.push(RenderedMarkdownLine {
816 line: Line::from(""),
817 links: Vec::new(),
818 is_code: false,
819 copy_prefix_width: 0,
820 copy_separator_after: CopyLineSeparator::Newline,
821 });
822 }
823
824 out
825 }
826
827 fn empty_rendered_markdown_line() -> RenderedMarkdownLine {
828 RenderedMarkdownLine {
829 line: Line::from(""),
830 links: Vec::new(),
831 is_code: false,
832 copy_prefix_width: 0,
833 copy_separator_after: CopyLineSeparator::Newline,
834 }
835 }
836
837 /// Render a block suffix while carrying syntax state across calls.
838 ///
839 /// Non-code blocks use the canonical batch renderer unchanged. Code lines are
840 /// the only group whose styling depends on preceding blocks, so their syntect
841 /// parse/highlight state is retained explicitly and cloned for the replaceable
842 /// tail. This keeps an open fence incremental without sacrificing exact final
843 /// highlighting.
844 fn render_incremental_blocks(
845 blocks: &[Block],
846 key: IncrementalRenderKey,
847 code_highlighter: &mut Option<IncrementalCodeHighlighter>,
848 ) -> Vec<RenderedMarkdownLine> {
849 let mut out = Vec::new();
850 let mut index = 0;
851 while index < blocks.len() {
852 if let Block::Code {
853 line,
854 language,
855 block_id,
856 } = &blocks[index]
857 {
858 let spans = highlight_incremental_code_line(
859 *block_id,
860 language.as_deref(),
861 line,
862 key.base_style,
863 key.palette_mode,
864 code_highlighter,
865 );
866 out.extend(render_wrapped_code_spans_tagged(
867 spans,
868 usize::from(key.width),
869 ));
870 index += 1;
871 continue;
872 }
873
874 let start = index;
875 while index < blocks.len() && !matches!(blocks[index], Block::Code { .. }) {
876 index += 1;
877 }
878 out.extend(render_parsed_tagged_with_palette(
879 &ParsedMarkdown {
880 blocks: blocks[start..index].to_vec(),
881 },
882 key.width,
883 key.base_style,
884 key.palette_mode,
885 ));
886 }
887 out
888 }
889
890 fn highlight_incremental_code_line(
891 block_id: usize,
892 language: Option<&str>,
893 line: &str,
894 base_style: Style,
895 palette_mode: palette::PaletteMode,
896 cache: &mut Option<IncrementalCodeHighlighter>,
897 ) -> Vec<Span<'static>> {
898 let language_owned = language.map(str::to_owned);
899 let needs_reset = cache
900 .as_ref()
901 .is_none_or(|current| current.block_id != block_id || current.language != language_owned);
902 if needs_reset {
903 let state = language
904 .and_then(find_code_syntax)
905 .map(|syntax| HighlightLines::new(syntax, selected_syntax_theme(palette_mode)).state());
906 *cache = Some(IncrementalCodeHighlighter {
907 block_id,
908 language: language_owned,
909 state,
910 });
911 }
912
913 let plain_style = base_style.fg(palette::TEXT_TOOL_OUTPUT);
914 let Some(current) = cache.as_mut() else {
915 return vec![Span::styled(line.to_string(), plain_style)];
916 };
917 let Some((highlight_state, parse_state)) = current.state.take() else {
918 return vec![Span::styled(line.to_string(), plain_style)];
919 };
920 let mut highlighter = HighlightLines::from_state(
921 selected_syntax_theme(palette_mode),
922 highlight_state,
923 parse_state,
924 );
925 let highlighted = match highlighter.highlight_line(line, syntax_set()) {
926 Ok(ranges) if !ranges.is_empty() => ranges
927 .into_iter()
928 .map(|(style, text)| {
929 Span::styled(
930 text.to_string(),
931 syntax_style_to_ratatui(style, base_style, palette_mode),
932 )
933 })
934 .collect(),
935 _ => vec![Span::styled(line.to_string(), plain_style)],
936 };
937 current.state = Some(highlighter.state());
938 highlighted
939 }
940
941 /// Convenience wrapper: parse + render in one call.
942 ///
943 /// Equivalent to `render_parsed(&parse(content), width, base_style)`. Callers
944 /// that don't manage their own cache (the Thinking body, the immediate message
945 /// body) use this.
946 #[must_use]
947 pub fn render_markdown(content: &str, width: u16, base_style: Style) -> Vec<Line<'static>> {
948 let parsed = parse(content);
949 render_parsed(&parsed, width, base_style)
950 }
951
952 /// Convenience wrapper: parse + render while keeping per-line source metadata.
953 #[cfg(test)]
954 #[must_use]
955 pub fn render_markdown_tagged(
956 content: &str,
957 width: u16,
958 base_style: Style,
959 ) -> Vec<RenderedMarkdownLine> {
960 let parsed = parse(content);
961 render_parsed_tagged(&parsed, width, base_style)
962 }
963
964 /// Parse and render markdown using an already-resolved UI palette mode.
965 #[must_use]
966 pub(crate) fn render_markdown_tagged_with_palette(
967 content: &str,
968 width: u16,
969 base_style: Style,
970 palette_mode: palette::PaletteMode,
971 ) -> Vec<RenderedMarkdownLine> {
972 let parsed = parse(content);
973 render_parsed_tagged_with_palette(&parsed, width, base_style, palette_mode)
974 }
975
976 /// Render plain text: split on newlines, word-wrap each line independently,
977 /// preserve leading whitespace and blank lines. No markdown interpretation.
978 #[must_use]
979 pub fn render_plain_text(content: &str, width: u16, base_style: Style) -> Vec<Line<'static>> {
980 let width = width.max(1) as usize;
981 let mut lines = Vec::new();
982 for raw_line in content.split('\n') {
983 if raw_line.is_empty() {
984 lines.push(Line::from(""));
985 } else {
986 lines.extend(wrap_plain_line(raw_line, width, base_style));
987 }
988 }
989 if lines.is_empty() {
990 lines.push(Line::from(""));
991 }
992 lines
993 }
994
995 /// Word-wrap a single line at `width`, preserving leading whitespace.
996 /// Handles over-long words by char-breaking (same strategy as the markdown
997 /// line renderer).
998 fn wrap_plain_line(line: &str, width: usize, style: Style) -> Vec<Line<'static>> {
999 if width == 0 || line.is_empty() {
1000 return vec![Line::from("")];
1001 }
1002
1003 let mut chunks = Vec::new();
1004 let mut current = String::new();
1005 let mut current_width = 0usize;
1006 let mut last_break_pos = None;
1007
1008 for grapheme in line.graphemes(true) {
1009 loop {
1010 let grapheme_width = markdown_grapheme_width(grapheme, current_width);
1011 if current_width + grapheme_width <= width || current.is_empty() {
1012 break;
1013 }
1014
1015 if let Some(pos) = last_break_pos {
1016 if pos == current.len() {
1017 chunks.push(std::mem::take(&mut current));
1018 current_width = 0;
1019 last_break_pos = None;
1020 break;
1021 }
1022
1023 if current[..pos].chars().any(|c| !c.is_whitespace()) {
1024 let tail = current.split_off(pos);
1025 chunks.push(std::mem::take(&mut current));
1026 current = tail;
1027 current_width = plain_display_width(&current);
1028 last_break_pos = last_plain_break_pos(&current);
1029 continue;
1030 }
1031 }
1032
1033 chunks.push(std::mem::take(&mut current));
1034 current_width = 0;
1035 last_break_pos = None;
1036 break;
1037 }
1038
1039 let grapheme_width = markdown_grapheme_width(grapheme, current_width);
1040 current.push_str(grapheme);
1041 current_width += grapheme_width;
1042 if grapheme.chars().all(char::is_whitespace) {
1043 last_break_pos = Some(current.len());
1044 }
1045 }
1046
1047 if !current.is_empty() {
1048 chunks.push(current);
1049 }
1050
1051 if chunks.is_empty() {
1052 return vec![Line::from("")];
1053 }
1054
1055 chunks
1056 .into_iter()
1057 .map(|chunk| Line::from(vec![Span::styled(chunk, style)]))
1058 .collect()
1059 }
1060
1061 fn plain_display_width(text: &str) -> usize {
1062 let mut width = 0usize;
1063 for grapheme in text.graphemes(true) {
1064 width += markdown_grapheme_width(grapheme, width);
1065 }
1066 width
1067 }
1068
1069 fn last_plain_break_pos(text: &str) -> Option<usize> {
1070 text.char_indices()
1071 .rev()
1072 .find_map(|(idx, ch)| ch.is_whitespace().then_some(idx + ch.len_utf8()))
1073 }
1074
1075 fn parse_heading(line: &str) -> Option<(usize, &str)> {
1076 let trimmed = line.trim_start();
1077 let hashes = trimmed.chars().take_while(|c| *c == '#').count();
1078 if hashes == 0 {
1079 return None;
1080 }
1081 let text = trimmed[hashes..].trim();
1082 if text.is_empty() {
1083 None
1084 } else {
1085 Some((hashes, text))
1086 }
1087 }
1088
1089 fn parse_list_item(line: &str) -> Option<(String, &str)> {
1090 let trimmed = line.trim_start();
1091 if trimmed.starts_with("- ") || trimmed.starts_with("* ") {
1092 return Some(("-".to_string(), trimmed[2..].trim()));
1093 }
1094 let bytes = trimmed.as_bytes();
1095 let mut idx = 0;
1096 while idx < bytes.len() && bytes[idx].is_ascii_digit() {
1097 idx += 1;
1098 }
1099 if idx == 0 || idx >= bytes.len() || bytes[idx] != b'.' {
1100 return None;
1101 }
1102 let rest = &trimmed[idx + 1..];
1103 if !rest.starts_with(' ') {
1104 return None;
1105 }
1106 Some((format!("{}.", &trimmed[..idx]), rest.trim_start()))
1107 }
1108
1109 /// Upper bound on the nesting depth rendered for a `>` quote. Deeper quotes
1110 /// are clamped and the extra markers dropped from the rendered text; capping
1111 /// stops a pathological input like `>>>>>>>>>>>> text` from consuming the
1112 /// whole line width in rails.
1113 const MAX_QUOTE_DEPTH: usize = 4;
1114
1115 /// Parse a `>` blockquote line, returning `(depth, text)`.
1116 ///
1117 /// CommonMark nests with `>>` or `> >`; we count every leading `>` regardless
1118 /// of interleaved spaces, then trim the remaining content. A lone `>` yields
1119 /// an empty quote line. Deliberately lenient about missing space after `>` so
1120 /// model output like `>note` still renders as a quote.
1121 fn parse_blockquote(line: &str) -> Option<(usize, &str)> {
1122 let trimmed = line.trim_start();
1123 if !trimmed.starts_with('>') {
1124 return None;
1125 }
1126 let mut rest = trimmed;
1127 let mut depth = 0usize;
1128 while rest.starts_with('>') {
1129 depth = depth.saturating_add(1);
1130 rest = rest[1..].trim_start_matches([' ', '\t']);
1131 }
1132 Some((depth.clamp(1, MAX_QUOTE_DEPTH), rest.trim()))
1133 }
1134
1135 fn normalized_fence_language(info: &str) -> Option<String> {
1136 let token = info
1137 .trim()
1138 .split(|ch: char| ch.is_whitespace() || ch == ',')
1139 .next()
1140 .unwrap_or("")
1141 .trim_matches(['{', '}', '.'])
1142 .to_ascii_lowercase();
1143 if token.is_empty() || matches!(token.as_str(), "text" | "txt" | "plain" | "plaintext") {
1144 return None;
1145 }
1146 let normalized = match token.as_str() {
1147 "rs" => "rust",
1148 "js" | "jsx" | "node" => "javascript",
1149 "ts" | "tsx" => "typescript",
1150 "py" => "python",
1151 "rb" => "ruby",
1152 "sh" | "shell" | "zsh" => "bash",
1153 "yml" => "yaml",
1154 "md" => "markdown",
1155 other => other,
1156 };
1157 Some(normalized.to_string())
1158 }
1159
1160 fn selected_syntax_theme(mode: palette::PaletteMode) -> &'static Theme {
1161 let themes = theme_set();
1162 let preferred = match mode {
1163 palette::PaletteMode::Dark | palette::PaletteMode::Grayscale => "base16-ocean.dark",
1164 palette::PaletteMode::Light => "InspiredGitHub",
1165 palette::PaletteMode::SolarizedLight => "Solarized (light)",
1166 };
1167 themes
1168 .themes
1169 .get(preferred)
1170 .or_else(|| themes.themes.values().next())
1171 .expect("syntect ships at least one default theme")
1172 }
1173
1174 fn syntax_style_to_ratatui(
1175 style: syntect::highlighting::Style,
1176 base_style: Style,
1177 palette_mode: palette::PaletteMode,
1178 ) -> Style {
1179 let fg = syntax_rgb_to_terminal_color(
1180 style.foreground.r,
1181 style.foreground.g,
1182 style.foreground.b,
1183 palette_mode,
1184 syntax_color_depth(),
1185 );
1186 let mut modifiers = Modifier::empty();
1187 if style.font_style.contains(FontStyle::BOLD) {
1188 modifiers |= Modifier::BOLD;
1189 }
1190 if style.font_style.contains(FontStyle::ITALIC) {
1191 modifiers |= Modifier::ITALIC;
1192 }
1193 if style.font_style.contains(FontStyle::UNDERLINE) {
1194 modifiers |= Modifier::UNDERLINED;
1195 }
1196 base_style.fg(fg).add_modifier(modifiers)
1197 }
1198
1199 fn syntax_rgb_to_terminal_color(
1200 r: u8,
1201 g: u8,
1202 b: u8,
1203 mode: palette::PaletteMode,
1204 depth: palette::ColorDepth,
1205 ) -> Color {
1206 let (r, g, b) = if mode == palette::PaletteMode::Grayscale {
1207 let luma =
1208 ((u32::from(r) * 299 + u32::from(g) * 587 + u32::from(b) * 114 + 500) / 1000) as u8;
1209 let readable = luma.clamp(96, 232);
1210 (readable, readable, readable)
1211 } else {
1212 (r, g, b)
1213 };
1214 let mut color = Color::Rgb(r, g, b);
1215 if matches!(
1216 color,
1217 reserved if reserved == palette::WHALE_HUMAN
1218 || reserved == palette::WHALE_LIVE
1219 || reserved == palette::WHALE_ACTION
1220 || reserved == palette::WHALE_ERROR
1221 ) {
1222 // Syntax colors are content, not brand/attention/work/danger state.
1223 // Shift exact collisions before terminal-depth reduction so the cell
1224 // cannot acquire a reserved semantic role in the color backend.
1225 color = Color::Rgb(r, g, b.saturating_add(1));
1226 }
1227 let color = palette::adapt_color(color, depth);
1228 let reserved = [
1229 palette::WHALE_HUMAN,
1230 palette::WHALE_LIVE,
1231 palette::WHALE_ACTION,
1232 palette::WHALE_ERROR,
1233 ]
1234 .map(|semantic| palette::adapt_color(semantic, depth));
1235 if !reserved.contains(&color) {
1236 return color;
1237 }
1238
1239 // Quantization can make distinct RGB values collide again. Walk a small,
1240 // deterministic neutral ramp until the terminal-level color no longer
1241 // impersonates one of the four reserved semantic lanes.
1242 for delta in [17_u8, 34, 51, 68, 85, 102, 119, 136] {
1243 let candidate = palette::adapt_color(
1244 Color::Rgb(
1245 r.wrapping_add(delta),
1246 g.wrapping_add(delta / 2),
1247 b.wrapping_add(delta / 3),
1248 ),
1249 depth,
1250 );
1251 if !reserved.contains(&candidate) {
1252 return candidate;
1253 }
1254 }
1255 // All supported depths have more than four colors, so this is only a
1256 // defensive fallback for a future adapter with a narrower gamut.
1257 Color::Reset
1258 }
1259
1260 fn highlight_code_block(
1261 language: Option<&str>,
1262 lines: &[&str],
1263 base_style: Style,
1264 palette_mode: palette::PaletteMode,
1265 ) -> Vec<Vec<Span<'static>>> {
1266 let plain_style = base_style.fg(palette::TEXT_TOOL_OUTPUT);
1267 let Some(language) = language else {
1268 return lines
1269 .iter()
1270 .map(|line| vec![Span::styled((*line).to_string(), plain_style)])
1271 .collect();
1272 };
1273 let syntaxes = syntax_set();
1274 let Some(syntax) = find_code_syntax(language) else {
1275 return lines
1276 .iter()
1277 .map(|line| vec![Span::styled((*line).to_string(), plain_style)])
1278 .collect();
1279 };
1280
1281 let mut highlighter = HighlightLines::new(syntax, selected_syntax_theme(palette_mode));
1282 lines
1283 .iter()
1284 .map(|line| match highlighter.highlight_line(line, syntaxes) {
1285 Ok(ranges) if !ranges.is_empty() => ranges
1286 .into_iter()
1287 .map(|(style, text)| {
1288 Span::styled(
1289 text.to_string(),
1290 syntax_style_to_ratatui(style, base_style, palette_mode),
1291 )
1292 })
1293 .collect(),
1294 _ => vec![Span::styled((*line).to_string(), plain_style)],
1295 })
1296 .collect()
1297 }
1298
1299 fn find_code_syntax(language: &str) -> Option<&'static syntect::parsing::SyntaxReference> {
1300 let syntaxes = syntax_set();
1301 syntaxes
1302 .find_syntax_by_token(language)
1303 .or_else(|| syntaxes.find_syntax_by_extension(language))
1304 .or_else(|| {
1305 syntaxes
1306 .syntaxes()
1307 .iter()
1308 .find(|syntax| syntax.name.eq_ignore_ascii_case(language))
1309 })
1310 }
1311
1312 fn render_wrapped_code_spans_tagged(
1313 spans: Vec<Span<'static>>,
1314 width: usize,
1315 ) -> Vec<RenderedMarkdownLine> {
1316 let prefix = " ";
1317 let prefix_width = prefix.width();
1318 let available = width.saturating_sub(prefix_width).max(1);
1319 let mut rows: Vec<Vec<(String, Style)>> = vec![Vec::new()];
1320 let mut current_width = 0usize;
1321
1322 for span in spans {
1323 for grapheme in span.content.graphemes(true) {
1324 let grapheme_width = markdown_grapheme_width(grapheme, current_width);
1325 if current_width + grapheme_width > available && current_width > 0 {
1326 rows.push(Vec::new());
1327 current_width = 0;
1328 }
1329 let row = rows.last_mut().expect("code rows are never empty");
1330 if let Some((text, style)) = row.last_mut()
1331 && *style == span.style
1332 {
1333 text.push_str(grapheme);
1334 } else {
1335 row.push((grapheme.to_string(), span.style));
1336 }
1337 current_width += markdown_grapheme_width(grapheme, current_width);
1338 }
1339 }
1340
1341 let last_index = rows.len().saturating_sub(1);
1342 rows.into_iter()
1343 .enumerate()
1344 .map(|(idx, row)| {
1345 let mut rendered = vec![Span::raw(prefix)];
1346 rendered.extend(
1347 row.into_iter()
1348 .map(|(text, style)| Span::styled(text, style)),
1349 );
1350 RenderedMarkdownLine {
1351 line: Line::from(rendered),
1352 links: Vec::new(),
1353 is_code: true,
1354 copy_prefix_width: prefix_width,
1355 copy_separator_after: if idx == last_index {
1356 CopyLineSeparator::Newline
1357 } else {
1358 CopyLineSeparator::None
1359 },
1360 }
1361 })
1362 .collect()
1363 }
1364
1365 fn render_wrapped_line_tagged(
1366 line: &str,
1367 width: usize,
1368 style: Style,
1369 indent_code: bool,
1370 is_code: bool,
1371 ) -> Vec<RenderedMarkdownLine> {
1372 let prefix = if indent_code { " " } else { "" };
1373 let prefix_width = prefix.width();
1374 let available = width.saturating_sub(prefix_width).max(1);
1375 // Code blocks must preserve leading whitespace (indentation is semantic).
1376 // Use hard character-width wrapping instead of word-wrap.
1377 let wrapped = if indent_code {
1378 wrap_code_line(line, available)
1379 } else {
1380 wrap_text(line, available)
1381 };
1382 let mut out = Vec::new();
1383
1384 let last_index = wrapped.len().saturating_sub(1);
1385 for (idx, chunk) in wrapped.into_iter().enumerate() {
1386 let line = if idx == 0 {
1387 Line::from(vec![Span::raw(prefix), Span::styled(chunk, style)])
1388 } else {
1389 Line::from(vec![
1390 Span::raw(" ".repeat(prefix_width)),
1391 Span::styled(chunk, style),
1392 ])
1393 };
1394 let copy_separator_after = if idx == last_index {
1395 CopyLineSeparator::Newline
1396 } else if is_code {
1397 CopyLineSeparator::None
1398 } else {
1399 CopyLineSeparator::Space
1400 };
1401 out.push(RenderedMarkdownLine {
1402 line,
1403 links: Vec::new(),
1404 is_code,
1405 copy_prefix_width: if indent_code { prefix_width } else { 0 },
1406 copy_separator_after,
1407 });
1408 }
1409
1410 out
1411 }
1412
1413 fn render_list_line_tagged(
1414 bullet: &str,
1415 text: &str,
1416 width: usize,
1417 bullet_style: Style,
1418 text_style: Style,
1419 ) -> Vec<RenderedMarkdownLine> {
1420 let bullet_prefix = format!("{bullet} ");
1421 let bullet_width = bullet_prefix.width();
1422 let available = width.saturating_sub(bullet_width).max(1);
1423 let wrapped = render_line_with_links_tagged(text, available, text_style, link_style());
1424
1425 let mut out = Vec::new();
1426 for (idx, rendered) in wrapped.into_iter().enumerate() {
1427 let links = rendered
1428 .links
1429 .iter()
1430 .map(|link| link.shifted(bullet_width))
1431 .collect();
1432 if idx == 0 {
1433 let mut spans = vec![Span::styled(bullet_prefix.clone(), bullet_style)];
1434 spans.extend(rendered.line.spans);
1435 out.push(RenderedMarkdownLine {
1436 line: Line::from(spans),
1437 links,
1438 is_code: false,
1439 copy_prefix_width: 0,
1440 copy_separator_after: rendered.copy_separator_after,
1441 });
1442 } else {
1443 let mut spans = vec![Span::raw(" ".repeat(bullet_width))];
1444 spans.extend(rendered.line.spans);
1445 out.push(RenderedMarkdownLine {
1446 line: Line::from(spans),
1447 links,
1448 is_code: false,
1449 copy_prefix_width: bullet_width,
1450 copy_separator_after: rendered.copy_separator_after,
1451 });
1452 }
1453 }
1454 out
1455 }
1456
1457 /// Render a `>` quote line: a vertical-rule rail per nesting depth plus the
1458 /// quote text with inline formatting (bold, code, links).
1459 ///
1460 /// The rail is display chrome, not source markup. On the first row the
1461 /// transcript rail-scan (`compute_rail_prefix_width`) already strips it along
1462 /// with the assistant glyph, so `copy_prefix_width` must be `0` there — like
1463 /// list items — or selection copy would strip the rail twice and lose quote
1464 /// text. Wrapped continuation rows have no rail scan (plain spaces), so they
1465 /// report the rail width so selection copy skips their alignment.
1466 fn render_quote_line_tagged(
1467 text: &str,
1468 depth: usize,
1469 width: usize,
1470 rail_style: Style,
1471 text_style: Style,
1472 ) -> Vec<RenderedMarkdownLine> {
1473 let depth = depth.clamp(1, MAX_QUOTE_DEPTH);
1474 let rail = "│ ".repeat(depth);
1475 let rail_width = rail.width();
1476 let available = width.saturating_sub(rail_width).max(1);
1477 let wrapped = render_line_with_links_tagged(text, available, text_style, link_style());
1478
1479 let mut out = Vec::new();
1480 for (idx, rendered) in wrapped.into_iter().enumerate() {
1481 let links = rendered
1482 .links
1483 .iter()
1484 .map(|link| link.shifted(rail_width))
1485 .collect();
1486 let mut spans = if idx == 0 {
1487 (0..depth).map(|_| Span::styled("│ ", rail_style)).collect()
1488 } else {
1489 vec![Span::raw(" ".repeat(rail_width))]
1490 };
1491 spans.extend(rendered.line.spans);
1492 out.push(RenderedMarkdownLine {
1493 line: Line::from(spans),
1494 links,
1495 is_code: false,
1496 // First row: the transcript rail-scan strips the visible rail
1497 // (mirror `render_list_line_tagged`); continuation rows: report
1498 // the alignment width so selection copy skips it.
1499 copy_prefix_width: if idx == 0 { 0 } else { rail_width },
1500 copy_separator_after: rendered.copy_separator_after,
1501 });
1502 }
1503 out
1504 }
1505
1506 #[cfg(test)]
1507 fn render_line_with_links(
1508 line: &str,
1509 width: usize,
1510 base_style: Style,
1511 link_style: Style,
1512 ) -> Vec<Line<'static>> {
1513 render_line_with_links_tagged(line, width, base_style, link_style)
1514 .into_iter()
1515 .map(|rendered| rendered.line)
1516 .collect()
1517 }
1518
1519 fn render_line_with_links_tagged(
1520 line: &str,
1521 width: usize,
1522 base_style: Style,
1523 link_style: Style,
1524 ) -> Vec<RenderedMarkdownLine> {
1525 if line.trim().is_empty() {
1526 return vec![RenderedMarkdownLine {
1527 line: Line::from(""),
1528 links: Vec::new(),
1529 is_code: false,
1530 copy_prefix_width: 0,
1531 copy_separator_after: CopyLineSeparator::Newline,
1532 }];
1533 }
1534
1535 // Flatten inline tokens into (word, style) pairs preserving inter-token spaces.
1536 let tokens = parse_inline_spans(line, base_style, link_style);
1537 let mut words: Vec<InlineToken> = Vec::new();
1538 for token in tokens {
1539 let mut first = true;
1540 for part in token.text.split(' ') {
1541 if !first {
1542 // The space consumed by split remains part of a markdown-link
1543 // label when the surrounding token is linked. It is still a
1544 // wrap opportunity and is dropped at a row boundary.
1545 words.push(InlineToken::new(
1546 " ".to_string(),
1547 token.style,
1548 token.link_url.clone(),
1549 ));
1550 }
1551 if !part.is_empty() {
1552 words.push(InlineToken::new(
1553 part.to_string(),
1554 token.style,
1555 token.link_url.clone(),
1556 ));
1557 }
1558 first = false;
1559 }
1560 }
1561
1562 let mut lines: Vec<RenderedMarkdownLine> = Vec::new();
1563 let mut current_spans: Vec<Span<'static>> = Vec::new();
1564 let mut current_links: Vec<osc8::LineLink> = Vec::new();
1565 let mut current_width = 0usize;
1566
1567 for word in words {
1568 let ww = word.text.width();
1569 if word.text == " " {
1570 // Space: emit only if we're mid-line and it fits; otherwise drop
1571 // (it's a potential wrap point, not content).
1572 if !current_spans.is_empty() && current_width < width {
1573 current_spans.push(word.span_for(" ".to_string()));
1574 record_inline_link(&mut current_links, &word, current_width, 1);
1575 current_width += 1;
1576 }
1577 continue;
1578 }
1579 // If the word itself is wider than an entire line, hard-break it at
1580 // grapheme boundaries so wrapping always makes progress (#1344,
1581 // #1351). Without this, long URLs / paths / hashes were placed on
1582 // their own line whole and silently overflowed the right edge of
1583 // the transcript.
1584 if ww > width && width > 0 {
1585 // Flush the in-progress line first.
1586 if !current_spans.is_empty() {
1587 push_inline_line(
1588 &mut lines,
1589 &mut current_spans,
1590 &mut current_links,
1591 CopyLineSeparator::Space,
1592 );
1593 current_width = 0;
1594 }
1595 // Char-break the word into width-sized chunks. Each full chunk
1596 // becomes its own line; the final partial chunk continues the
1597 // current line so the next word can pack onto it.
1598 let mut chunk = String::new();
1599 let mut chunk_w = 0usize;
1600 for grapheme in word.text.graphemes(true) {
1601 let grapheme_width = grapheme.width();
1602 if chunk_w + grapheme_width > width && chunk_w > 0 {
1603 let chunk = std::mem::take(&mut chunk);
1604 let mut links = Vec::new();
1605 record_inline_link(&mut links, &word, 0, chunk_w);
1606 lines.push(RenderedMarkdownLine {
1607 line: Line::from(vec![word.span_for(chunk)]),
1608 links,
1609 is_code: false,
1610 copy_prefix_width: 0,
1611 copy_separator_after: CopyLineSeparator::None,
1612 });
1613 chunk_w = 0;
1614 }
1615 chunk.push_str(grapheme);
1616 chunk_w += grapheme_width;
1617 }
1618 if !chunk.is_empty() {
1619 record_inline_link(&mut current_links, &word, 0, chunk_w);
1620 current_spans.push(word.span_for(chunk));
1621 current_width = chunk_w;
1622 }
1623 continue;
1624 }
1625 // Wrap before this word if it doesn't fit.
1626 if current_width > 0 && current_width + ww > width {
1627 // Trim trailing space span before breaking.
1628 push_inline_line(
1629 &mut lines,
1630 &mut current_spans,
1631 &mut current_links,
1632 CopyLineSeparator::Space,
1633 );
1634 current_width = 0;
1635 }
1636 record_inline_link(&mut current_links, &word, current_width, ww);
1637 current_spans.push(word.into_span());
1638 current_width += ww;
1639 }
1640
1641 if !current_spans.is_empty() {
1642 push_inline_line(
1643 &mut lines,
1644 &mut current_spans,
1645 &mut current_links,
1646 CopyLineSeparator::Newline,
1647 );
1648 } else if let Some(last) = lines.last_mut() {
1649 last.copy_separator_after = CopyLineSeparator::Newline;
1650 }
1651 if lines.is_empty() {
1652 lines.push(RenderedMarkdownLine {
1653 line: Line::from(""),
1654 links: Vec::new(),
1655 is_code: false,
1656 copy_prefix_width: 0,
1657 copy_separator_after: CopyLineSeparator::Newline,
1658 });
1659 }
1660 lines
1661 }
1662
1663 fn push_inline_line(
1664 lines: &mut Vec<RenderedMarkdownLine>,
1665 spans: &mut Vec<Span<'static>>,
1666 links: &mut Vec<osc8::LineLink>,
1667 copy_separator_after: CopyLineSeparator,
1668 ) {
1669 if let Some(last) = spans.last()
1670 && last.content.as_ref() == " "
1671 {
1672 spans.pop();
1673 }
1674 let visible_width = spans
1675 .iter()
1676 .map(|span| span.content.as_ref().width())
1677 .sum::<usize>();
1678 links.retain(|link| link.col_start < visible_width);
1679 for link in links.iter_mut() {
1680 link.col_end = link.col_end.min(visible_width.saturating_sub(1));
1681 }
1682 lines.push(RenderedMarkdownLine {
1683 line: Line::from(std::mem::take(spans)),
1684 links: std::mem::take(links),
1685 is_code: false,
1686 copy_prefix_width: 0,
1687 copy_separator_after,
1688 });
1689 }
1690
1691 fn record_inline_link(
1692 links: &mut Vec<osc8::LineLink>,
1693 token: &InlineToken,
1694 col_start: usize,
1695 width: usize,
1696 ) {
1697 let Some(target) = token.link_url.as_ref() else {
1698 return;
1699 };
1700 if width == 0 {
1701 return;
1702 }
1703 let col_end = col_start.saturating_add(width).saturating_sub(1);
1704 if let Some(last) = links.last_mut()
1705 && last.target == *target
1706 && last.col_end.saturating_add(1) == col_start
1707 {
1708 last.col_end = col_end;
1709 return;
1710 }
1711 links.push(osc8::LineLink {
1712 col_start,
1713 col_end,
1714 target: target.clone(),
1715 });
1716 }
1717
1718 #[derive(Clone)]
1719 struct InlineToken {
1720 text: String,
1721 style: Style,
1722 link_url: Option<String>,
1723 }
1724
1725 impl InlineToken {
1726 fn new(text: String, style: Style, link_url: Option<String>) -> Self {
1727 Self {
1728 text,
1729 style,
1730 link_url,
1731 }
1732 }
1733
1734 fn span_for(&self, text: String) -> Span<'static> {
1735 Span::styled(text, self.style)
1736 }
1737
1738 fn into_span(self) -> Span<'static> {
1739 Span::styled(self.text, self.style)
1740 }
1741 }
1742
1743 /// Parse an entire line into (text, style) segments, handling **bold**,
1744 /// *italic*, `code`, ~~strikethrough~~, `[text](url)` links, and bare URLs.
1745 fn parse_inline_spans(line: &str, base_style: Style, link_style: Style) -> Vec<InlineToken> {
1746 let bold_style = base_style.add_modifier(Modifier::BOLD);
1747 let italic_style = base_style.add_modifier(Modifier::ITALIC);
1748 let code_style = base_style
1749 .add_modifier(Modifier::ITALIC)
1750 .bg(palette::SURFACE_ELEVATED);
1751 let strike_style = base_style.add_modifier(Modifier::CROSSED_OUT);
1752 let mut out = Vec::new();
1753 let mut rest = line;
1754
1755 while !rest.is_empty() {
1756 // Backslash escape (CommonMark §2.4): `\` before ASCII punctuation
1757 // yields the literal character. Producers escape untrusted text
1758 // (plugin names, paths) this way so it cannot open markup; without
1759 // this arm the backslash itself reached the terminal
1760 // (`computer\-use`, `0\.1\.0` — 0.9.12 defect #21).
1761 if let Some(escaped) = rest.strip_prefix('\\')
1762 && let Some(ch) = escaped.chars().next()
1763 && ch.is_ascii_punctuation()
1764 {
1765 out.push(InlineToken::new(ch.to_string(), base_style, None));
1766 rest = &escaped[ch.len_utf8()..];
1767 continue;
1768 }
1769 // **bold**
1770 if let Some(end) = rest.strip_prefix("**").and_then(|s| s.find("**")) {
1771 let inner = &rest[2..2 + end];
1772 out.push(InlineToken::new(inner.to_string(), bold_style, None));
1773 rest = &rest[2 + end + 2..];
1774 continue;
1775 }
1776 // __bold__
1777 if let Some(end) = rest.strip_prefix("__").and_then(|s| s.find("__")) {
1778 let inner = &rest[2..2 + end];
1779 out.push(InlineToken::new(inner.to_string(), bold_style, None));
1780 rest = &rest[2 + end + 2..];
1781 continue;
1782 }
1783 // *italic*
1784 if rest.starts_with('*')
1785 && !rest.starts_with("**")
1786 && let Some(end) = rest[1..].find('*')
1787 {
1788 let inner = &rest[1..1 + end];
1789 let after = &rest[1 + end + 1..];
1790 // Closing delimiter must not be immediately followed by a
1791 // letter, digit, or underscore (otherwise it's part of an
1792 // identifier like `codewhale_tui`, not italic markup).
1793 if !after.starts_with(|c: char| c.is_alphanumeric() || c == '_') {
1794 out.push(InlineToken::new(inner.to_string(), italic_style, None));
1795 rest = after;
1796 continue;
1797 }
1798 }
1799 // _italic_
1800 if rest.starts_with('_')
1801 && !rest.starts_with("__")
1802 && let Some(end) = rest[1..].find('_')
1803 {
1804 let inner = &rest[1..1 + end];
1805 let after = &rest[1 + end + 1..];
1806 // CommonMark forbids an intraword `_` from opening emphasis and
1807 // requires the opener to be left-flanking (not followed by
1808 // whitespace). Guarding only the closer let `b_p … t_?` italicize
1809 // the prose between two math subscripts (#6042).
1810 let preceded_by_word = line[..line.len() - rest.len()]
1811 .chars()
1812 .next_back()
1813 .is_some_and(char::is_alphanumeric);
1814 let openable =
1815 !preceded_by_word && inner.chars().next().is_some_and(|c| !c.is_whitespace());
1816 // Closing delimiter must not be immediately followed by a
1817 // letter, digit, or underscore.
1818 if openable && !after.starts_with(|c: char| c.is_alphanumeric() || c == '_') {
1819 out.push(InlineToken::new(inner.to_string(), italic_style, None));
1820 rest = after;
1821 continue;
1822 }
1823 }
1824 // `inline code`
1825 if let Some(end) = rest.strip_prefix('`').and_then(|s| s.find('`')) {
1826 let inner = &rest[1..1 + end];
1827 out.push(InlineToken::new(inner.to_string(), code_style, None));
1828 rest = &rest[1 + end + 1..];
1829 continue;
1830 }
1831 // ~~strikethrough~~
1832 if let Some(end) = rest.strip_prefix("~~").and_then(|s| s.find("~~")) {
1833 let inner = &rest[2..2 + end];
1834 out.push(InlineToken::new(inner.to_string(), strike_style, None));
1835 rest = &rest[2 + end + 2..];
1836 continue;
1837 }
1838 // [text](url)
1839 if rest.starts_with('[')
1840 && let Some(bracket_end) = rest.find(']')
1841 {
1842 let text = &rest[1..bracket_end];
1843 let after_bracket = &rest[bracket_end + 1..];
1844 if after_bracket.starts_with('(')
1845 && let Some(paren_end) = after_bracket.find(')')
1846 {
1847 let url = &after_bracket[1..paren_end];
1848 // The runtime toggle gates backend emission, not layout.
1849 // Keeping the same visible label and metadata in both modes
1850 // prevents toggling OSC 8 from reflowing the transcript.
1851 out.push(InlineToken::new(
1852 text.to_string(),
1853 link_style,
1854 normalized_link_target(url),
1855 ));
1856 rest = &after_bracket[paren_end + 1..];
1857 continue;
1858 }
1859 }
1860 // URL: consume until whitespace, then keep trailing punctuation
1861 // visible but outside the hyperlink target.
1862 if rest.starts_with("http://") || rest.starts_with("https://") {
1863 let token_end = rest.find(char::is_whitespace).unwrap_or(rest.len());
1864 let token = &rest[..token_end];
1865 let url_end = trailing_url_end(token);
1866 let url = &token[..url_end];
1867 out.push(InlineToken::new(
1868 url.to_string(),
1869 link_style,
1870 normalized_http_link_target(url),
1871 ));
1872 if url_end < token_end {
1873 out.push(InlineToken::new(
1874 token[url_end..].to_string(),
1875 base_style,
1876 None,
1877 ));
1878 }
1879 rest = &rest[token_end..];
1880 continue;
1881 }
1882 // Plain text: consume until next marker or URL; always advance at least 1 char.
1883 let next = find_next_marker(rest).max(rest.chars().next().map_or(1, |c| c.len_utf8()));
1884 out.push(InlineToken::new(rest[..next].to_string(), base_style, None));
1885 rest = &rest[next..];
1886 }
1887 out
1888 }
1889
1890 /// Normalize an explicit markdown link destination, the only construct the
1891 /// renderer treats as a structured reference. Prose is never scanned for
1892 /// path-shaped text: that linkifies identifiers and version strings, and points
1893 /// at files that do not exist.
1894 fn normalized_link_target(target: &str) -> Option<String> {
1895 normalized_http_link_target(target).or_else(|| normalized_file_link_target(target))
1896 }
1897
1898 /// A destination that is already an absolute path, with or without the `file:`
1899 /// scheme. Relative destinations stay inert because this layer has no workspace
1900 /// root to resolve them against, and a `file://host/...` form is rejected rather
1901 /// than reinterpreted as local. Windows paths must therefore arrive pre-formed
1902 /// as `file:///C:/…`.
1903 fn normalized_file_link_target(target: &str) -> Option<String> {
1904 let path = match target.get(..7) {
1905 Some(prefix) if prefix.eq_ignore_ascii_case("file://") => &target[7..],
1906 _ => target,
1907 };
1908 if !path.starts_with('/') || path.chars().any(|ch| ch.is_whitespace() || ch.is_control()) {
1909 return None;
1910 }
1911 Some(format!("file://{path}"))
1912 }
1913
1914 /// OSC 8 targets produced by markdown are deliberately limited to ordinary
1915 /// web URLs. The browser-opening gesture is user-initiated, but accepting
1916 /// arbitrary schemes here would still turn untrusted model output into a
1917 /// `javascript:` or application-protocol link. Normalize the scheme and reject
1918 /// whitespace/control characters before metadata reaches a frame.
1919 fn normalized_http_link_target(target: &str) -> Option<String> {
1920 let (scheme, rest) = if target
1921 .get(..8)
1922 .is_some_and(|prefix| prefix.eq_ignore_ascii_case("https://"))
1923 {
1924 ("https://", &target[8..])
1925 } else if target
1926 .get(..7)
1927 .is_some_and(|prefix| prefix.eq_ignore_ascii_case("http://"))
1928 {
1929 ("http://", &target[7..])
1930 } else {
1931 return None;
1932 };
1933 if rest.is_empty()
1934 || rest.chars().any(|ch| ch.is_whitespace() || ch.is_control())
1935 || rest.split(['/', '?', '#']).next().is_none_or(str::is_empty)
1936 {
1937 return None;
1938 }
1939 Some(format!("{scheme}{rest}"))
1940 }
1941
1942 fn trailing_url_end(candidate: &str) -> usize {
1943 let mut end = candidate.len();
1944 while end > 0 {
1945 let remaining = &candidate[..end];
1946 let Some(ch) = remaining.chars().next_back() else {
1947 break;
1948 };
1949 let trim = matches!(ch, ',' | '.' | ';' | '!' | '\'' | '"')
1950 || matches!(ch, ')' | ']' | '}' | '>')
1951 && has_unmatched_closing_delimiter(remaining, ch);
1952 if !trim {
1953 break;
1954 }
1955 end -= ch.len_utf8();
1956 }
1957 end
1958 }
1959
1960 fn has_unmatched_closing_delimiter(candidate: &str, closing: char) -> bool {
1961 let opening = match closing {
1962 ')' => '(',
1963 ']' => '[',
1964 '}' => '{',
1965 '>' => '<',
1966 _ => return false,
1967 };
1968 candidate.chars().filter(|ch| *ch == closing).count()
1969 > candidate.chars().filter(|ch| *ch == opening).count()
1970 }
1971
1972 /// Find the index of the next inline marker (`**`, `__`, `*`, `_`, `http`)
1973 /// in `s`, or `s.len()` if none found.
1974 fn find_next_marker(s: &str) -> usize {
1975 let mut i = 0;
1976 let bytes = s.as_bytes();
1977 while i < bytes.len() {
1978 let ch_len = s[i..].chars().next().map_or(1, |c| c.len_utf8());
1979 let slice = &s[i..];
1980 if slice.starts_with('\\')
1981 || slice.starts_with("**")
1982 || slice.starts_with("__")
1983 || slice.starts_with("~~")
1984 || slice.starts_with('`')
1985 || slice.starts_with('[')
1986 || (slice.starts_with('*') && !slice.starts_with("**"))
1987 || (slice.starts_with('_') && !slice.starts_with("__"))
1988 || slice.starts_with("http://")
1989 || slice.starts_with("https://")
1990 {
1991 return i;
1992 }
1993 i += ch_len;
1994 }
1995 s.len()
1996 }
1997
1998 fn is_horizontal_rule(line: &str) -> bool {
1999 let stripped: String = line.chars().filter(|c| !c.is_whitespace()).collect();
2000 (stripped.chars().all(|c| c == '-')
2001 || stripped.chars().all(|c| c == '*')
2002 || stripped.chars().all(|c| c == '_'))
2003 && stripped.len() >= 3
2004 }
2005
2006 /// Parse a markdown table row like `| foo | bar |` into trimmed cell strings.
2007 /// Returns `None` for separator rows (`|---|---|`).
2008 fn parse_table_row(line: &str) -> Option<Vec<String>> {
2009 if !line.starts_with('|') {
2010 return None;
2011 }
2012 let inner = line.trim_matches('|');
2013 let cells = split_table_cells(inner);
2014 // Separator row: every non-empty cell is only dashes/colons/spaces
2015 if cells
2016 .iter()
2017 .all(|c| c.is_empty() || c.chars().all(|ch| ch == '-' || ch == ':' || ch == ' '))
2018 {
2019 return None;
2020 }
2021 Some(cells)
2022 }
2023
2024 fn split_table_cells(inner: &str) -> Vec<String> {
2025 let mut cells = Vec::new();
2026 let mut current = String::new();
2027 let mut in_code = false;
2028 let mut chars = inner.chars().peekable();
2029
2030 while let Some(ch) = chars.next() {
2031 match ch {
2032 '\\' => {
2033 if matches!(chars.peek(), Some('|')) {
2034 current.push('|');
2035 let _ = chars.next();
2036 } else {
2037 current.push(ch);
2038 }
2039 }
2040 '`' => {
2041 in_code = !in_code;
2042 current.push(ch);
2043 }
2044 '|' if !in_code => {
2045 cells.push(current.trim().to_string());
2046 current.clear();
2047 }
2048 _ => current.push(ch),
2049 }
2050 }
2051
2052 cells.push(current.trim().to_string());
2053 cells
2054 }
2055
2056 /// Word-wrap a single cell's text into one or more visual lines, each
2057 /// constrained to `col_width` display columns. Whitespace is the preferred
2058 /// break point; words wider than `col_width` are hard-broken at grapheme
2059 /// boundaries so wrapping always makes progress (no infinite loop on URLs
2060 /// or paths). Returns at least one segment.
2061 fn wrap_cell_text(cell: &str, col_width: usize) -> Vec<String> {
2062 if cell.is_empty() || cell.width() <= col_width {
2063 return vec![cell.to_string()];
2064 }
2065 let mut lines: Vec<String> = Vec::new();
2066 let mut current = String::new();
2067 let mut current_w = 0usize;
2068
2069 for word in cell.split_whitespace() {
2070 let word_w = word.width();
2071 if current_w == 0 {
2072 if word_w > col_width {
2073 push_word_breaking_graphemes(
2074 word,
2075 col_width,
2076 &mut current,
2077 &mut current_w,
2078 &mut lines,
2079 );
2080 } else {
2081 current.push_str(word);
2082 current_w = word_w;
2083 }
2084 } else if current_w + 1 + word_w <= col_width {
2085 current.push(' ');
2086 current.push_str(word);
2087 current_w += 1 + word_w;
2088 } else {
2089 lines.push(std::mem::take(&mut current));
2090 current_w = 0;
2091 if word_w > col_width {
2092 push_word_breaking_graphemes(
2093 word,
2094 col_width,
2095 &mut current,
2096 &mut current_w,
2097 &mut lines,
2098 );
2099 } else {
2100 current.push_str(word);
2101 current_w = word_w;
2102 }
2103 }
2104 }
2105 if !current.is_empty() || lines.is_empty() {
2106 lines.push(current);
2107 }
2108 lines
2109 }
2110
2111 fn render_table_row(cells: &[String], width: usize, base_style: Style) -> Vec<Line<'static>> {
2112 if cells.is_empty() {
2113 return vec![Line::from("")];
2114 }
2115 let col_width = (width.saturating_sub(3 * cells.len() + 1)) / cells.len();
2116 let col_width = col_width.max(4);
2117 let sep_style = Style::default().fg(palette::TEXT_DIM);
2118
2119 // Wrap each cell into one or more visual segments. The row's visual
2120 // height equals the tallest column. Cells that wrap to fewer segments
2121 // get blank-padded continuation lines so column separators stay aligned.
2122 let wrapped: Vec<Vec<String>> = cells.iter().map(|c| wrap_cell_text(c, col_width)).collect();
2123 let row_height = wrapped.iter().map(Vec::len).max().unwrap_or(1).max(1);
2124
2125 let mut lines: Vec<Line<'static>> = Vec::with_capacity(row_height);
2126 for row in 0..row_height {
2127 let mut spans: Vec<Span> = vec![Span::styled("│ ".to_string(), sep_style)];
2128 for (i, cell_segments) in wrapped.iter().enumerate() {
2129 let segment = cell_segments.get(row).map(String::as_str).unwrap_or("");
2130 let cell_spans = parse_inline_spans(segment, base_style, link_style());
2131 let cell_width: usize = cell_spans.iter().map(|token| token.text.width()).sum();
2132 let pad = col_width.saturating_sub(cell_width);
2133 for token in cell_spans {
2134 spans.push(token.into_span());
2135 }
2136 spans.push(Span::raw(" ".repeat(pad)));
2137 if i + 1 < cells.len() {
2138 spans.push(Span::styled(" │ ".to_string(), sep_style));
2139 } else {
2140 spans.push(Span::styled(" │".to_string(), sep_style));
2141 }
2142 }
2143 lines.push(Line::from(spans));
2144 }
2145 lines
2146 }
2147
2148 fn table_col_width(num_cols: usize, term_width: usize) -> usize {
2149 let col_width = (term_width.saturating_sub(3 * num_cols + 1)) / num_cols;
2150 col_width.max(4)
2151 }
2152
2153 fn render_table_border(
2154 num_cols: usize,
2155 col_width: usize,
2156 sep_style: Style,
2157 left: &str,
2158 mid: &str,
2159 right: &str,
2160 ) -> Line<'static> {
2161 let fill = "\u{2500}".repeat(col_width);
2162 let mut s = String::new();
2163 s.push_str(left);
2164 for i in 0..num_cols {
2165 s.push_str(&fill);
2166 if i + 1 < num_cols {
2167 s.push_str(mid);
2168 } else {
2169 s.push_str(right);
2170 }
2171 }
2172 Line::from(Span::styled(s, sep_style))
2173 }
2174
2175 fn render_table_group(blocks: &[Block], width: usize, base_style: Style) -> Vec<Line<'static>> {
2176 let sep_style = Style::default().fg(palette::TEXT_DIM);
2177
2178 let num_cols = blocks
2179 .iter()
2180 .filter_map(|b| match b {
2181 Block::TableRow(cells) => Some(cells.len()),
2182 _ => None,
2183 })
2184 .max()
2185 .unwrap_or(1);
2186
2187 let col_width = table_col_width(num_cols, width);
2188
2189 let mut lines = Vec::new();
2190
2191 // Top border
2192 lines.push(render_table_border(
2193 num_cols,
2194 col_width,
2195 sep_style,
2196 "\u{250C}\u{2500}",
2197 "\u{2500}\u{252C}\u{2500}",
2198 "\u{2500}\u{2510}",
2199 ));
2200
2201 let mid_border = || {
2202 render_table_border(
2203 num_cols,
2204 col_width,
2205 sep_style,
2206 "\u{251C}\u{2500}",
2207 "\u{2500}\u{253C}\u{2500}",
2208 "\u{2500}\u{2524}",
2209 )
2210 };
2211
2212 for i in 0..blocks.len() {
2213 match &blocks[i] {
2214 Block::TableRow(cells) => {
2215 lines.extend(render_table_row(cells, width, base_style));
2216 if i + 1 < blocks.len() && matches!(&blocks[i + 1], Block::TableRow(_)) {
2217 lines.push(mid_border());
2218 }
2219 }
2220 Block::TableSeparator => {
2221 lines.push(mid_border());
2222 }
2223 _ => {}
2224 }
2225 }
2226
2227 // Bottom border
2228 lines.push(render_table_border(
2229 num_cols,
2230 col_width,
2231 sep_style,
2232 "\u{2514}\u{2500}",
2233 "\u{2500}\u{2534}\u{2500}",
2234 "\u{2500}\u{2518}",
2235 ));
2236
2237 lines
2238 }
2239
2240 fn link_style() -> Style {
2241 Style::default()
2242 .fg(palette::WHALE_ACTION)
2243 .add_modifier(Modifier::UNDERLINED)
2244 }
2245
2246 /// Display-column width of one extended grapheme for terminal line-wrap
2247 /// calculations.
2248 ///
2249 /// A tab advances to the next 8-column tab stop. Single-codepoint characters
2250 /// retain the previous one-column fallback (with an override for enclosed
2251 /// alphanumerics that render as 2 columns in CJK terminals). Multi-codepoint
2252 /// emoji, keycaps, and combining sequences use the same string-level width
2253 /// contract as Ratatui, with an override for keycap sequences containing
2254 /// U+20E3 that unicode-width undercounts. (#4479)
2255 fn markdown_grapheme_width(grapheme: &str, col: usize) -> usize {
2256 if grapheme == "\t" {
2257 return 8 - (col % 8); // advance to next 8-column tab stop
2258 }
2259 if let Some(ch) = grapheme.chars().next()
2260 && ch.len_utf8() == grapheme.len()
2261 {
2262 return match ch {
2263 // Enclosed alphanumerics, dingbat circled digits, and circled
2264 // numbers on black square render as 2 columns in CJK terminals
2265 // even though unicode-width reports 1. (#4479)
2266 '\u{2460}'..='\u{24FF}' | '\u{2776}'..='\u{2793}' | '\u{3248}'..='\u{324F}' => 2,
2267 _ => ch.width().unwrap_or(1),
2268 };
2269 }
2270 // Keycap sequences (with or without FE0F) render as 2 columns.
2271 if grapheme.contains('\u{20e3}') {
2272 return 2;
2273 }
2274 grapheme.width()
2275 }
2276
2277 /// Hard-wrap a code line at `width` display columns, preserving all
2278 /// whitespace (including leading indentation). Unlike [`wrap_text`], this
2279 /// does not split on word boundaries — code indentation is semantic.
2280 fn wrap_code_line(line: &str, width: usize) -> Vec<String> {
2281 if width == 0 || line.is_empty() {
2282 return vec![line.to_string()];
2283 }
2284 let mut chunks = Vec::new();
2285 let mut current = String::new();
2286 let mut current_width = 0usize;
2287
2288 for grapheme in line.graphemes(true) {
2289 let grapheme_width = markdown_grapheme_width(grapheme, current_width);
2290 if current_width + grapheme_width > width && !current.is_empty() {
2291 chunks.push(current);
2292 current = String::new();
2293 current_width = 0;
2294 }
2295 current.push_str(grapheme);
2296 current_width += grapheme_width;
2297 }
2298 chunks.push(current);
2299 chunks
2300 }
2301
2302 fn wrap_text(text: &str, width: usize) -> Vec<String> {
2303 if width == 0 {
2304 return vec![text.to_string()];
2305 }
2306 let mut lines = Vec::new();
2307 let mut current = String::new();
2308 let mut current_width = 0;
2309
2310 for word in text.split_whitespace() {
2311 let word_width = word.width();
2312 // If this single word is wider than the entire line, hard-break it
2313 // at grapheme boundaries so wrapping always makes progress
2314 // (#1344, #1351). Without this, long URLs / paths / hashes overflow
2315 // the right edge silently.
2316 if word_width > width {
2317 if !current.is_empty() {
2318 lines.push(std::mem::take(&mut current));
2319 current_width = 0;
2320 }
2321 push_word_breaking_graphemes(word, width, &mut current, &mut current_width, &mut lines);
2322 continue;
2323 }
2324 let additional = if current.is_empty() {
2325 word_width
2326 } else {
2327 word_width + 1
2328 };
2329 if current_width + additional > width && !current.is_empty() {
2330 lines.push(current);
2331 current = word.to_string();
2332 current_width = word_width;
2333 } else {
2334 if !current.is_empty() {
2335 current.push(' ');
2336 current_width += 1;
2337 }
2338 current.push_str(word);
2339 current_width += word_width;
2340 }
2341 }
2342
2343 if current.is_empty() {
2344 lines.push(String::new());
2345 } else {
2346 lines.push(current);
2347 }
2348
2349 lines
2350 }
2351
2352 /// Push graphemes from `word` into `current`, flushing to `lines` when the
2353 /// running display width would exceed `width`. String-level Unicode width
2354 /// matches Ratatui for emoji and combining sequences.
2355 /// Used by `wrap_text` and `wrap_cell_text` so a word longer than the
2356 /// allotted width never silently overflows the right edge.
2357 fn push_word_breaking_graphemes(
2358 word: &str,
2359 width: usize,
2360 current: &mut String,
2361 current_width: &mut usize,
2362 lines: &mut Vec<String>,
2363 ) {
2364 for grapheme in word.graphemes(true) {
2365 let grapheme_width = grapheme.width();
2366 if *current_width + grapheme_width > width && *current_width > 0 {
2367 lines.push(std::mem::take(current));
2368 *current_width = 0;
2369 }
2370 current.push_str(grapheme);
2371 *current_width += grapheme_width;
2372 }
2373 }
2374
2375 #[cfg(test)]
2376 mod tests {
2377 use super::*;
2378 use ratatui::style::Style;
2379
2380 fn visible_lines(lines: &[Line<'static>]) -> Vec<String> {
2381 lines
2382 .iter()
2383 .map(|line| {
2384 line.spans
2385 .iter()
2386 .map(|span| span.content.as_ref())
2387 .collect()
2388 })
2389 .collect()
2390 }
2391
2392 #[test]
2393 fn backslash_escapes_yield_the_literal_punctuation() {
2394 let lines = render_markdown(
2395 "ID: computer\\-use 0\\.1\\.0 \\*not italic\\* trailing\\",
2396 80,
2397 Style::default(),
2398 );
2399 assert_eq!(
2400 visible_lines(&lines),
2401 vec!["ID: computer-use 0.1.0 *not italic* trailing\\"]
2402 );
2403 }
2404
2405 fn rendered_fingerprint(lines: &[RenderedMarkdownLine]) -> Vec<String> {
2406 lines
2407 .iter()
2408 .map(|line| {
2409 format!(
2410 "{:?}|{:?}|{}|{}|{:?}",
2411 line.line,
2412 line.links,
2413 line.is_code,
2414 line.copy_prefix_width,
2415 line.copy_separator_after
2416 )
2417 })
2418 .collect()
2419 }
2420
2421 fn update_incremental_render(
2422 cache: &mut IncrementalMarkdownRenderCache,
2423 rendered: &mut Vec<RenderedMarkdownLine>,
2424 source: &str,
2425 width: u16,
2426 palette_mode: palette::PaletteMode,
2427 verified_append: bool,
2428 ) {
2429 update_incremental_render_with_style(
2430 cache,
2431 rendered,
2432 source,
2433 width,
2434 Style::default(),
2435 palette_mode,
2436 verified_append,
2437 );
2438 }
2439
2440 fn update_incremental_render_with_style(
2441 cache: &mut IncrementalMarkdownRenderCache,
2442 rendered: &mut Vec<RenderedMarkdownLine>,
2443 source: &str,
2444 width: u16,
2445 base_style: Style,
2446 palette_mode: palette::PaletteMode,
2447 verified_append: bool,
2448 ) {
2449 let delta = cache.update(source, width, base_style, palette_mode, verified_append);
2450 rendered.truncate(delta.replace_from);
2451 rendered.extend(delta.lines);
2452 }
2453
2454 #[test]
2455 fn incremental_render_is_exact_and_linear_for_unicode_fences_and_tables() {
2456 let mut cache = IncrementalMarkdownRenderCache::default();
2457 let mut rendered = Vec::new();
2458 let mut source = String::new();
2459 let chunks = 80usize;
2460
2461 for index in 0..chunks {
2462 source.push_str(&format!(
2463 "## 段落 {index}\nUnicode e\u{301} 世界 🚀\n```rust\nlet 値_{index}: usize = {index}; // 注釈\n```\n| key | value |\n|---|---|\n| {index} | 世界 |\n\n"
2464 ));
2465 update_incremental_render(
2466 &mut cache,
2467 &mut rendered,
2468 &source,
2469 96,
2470 palette::PaletteMode::Dark,
2471 index > 0,
2472 );
2473 let cold = render_markdown_tagged_with_palette(
2474 &source,
2475 96,
2476 Style::default(),
2477 palette::PaletteMode::Dark,
2478 );
2479 assert_eq!(
2480 rendered_fingerprint(&rendered),
2481 rendered_fingerprint(&cold),
2482 "incremental output diverged after chunk {index}"
2483 );
2484 }
2485
2486 let work = cache.work();
2487 let parsed = reference_parse(&source);
2488 assert_eq!(work.classified_lines as usize, source.lines().count());
2489 assert_eq!(work.stable_blocks_rendered as usize, parsed.blocks.len());
2490 assert_eq!(work.tail_blocks_rendered, 0);
2491 assert_eq!(work.invalidations, 1);
2492 }
2493
2494 #[test]
2495 fn incremental_render_invalidates_on_mutation_width_theme_and_style() {
2496 let mut cache = IncrementalMarkdownRenderCache::default();
2497 let mut rendered = Vec::new();
2498 let mut source = "alpha\n```rust\nlet value = 1;\n```\n".to_string();
2499 update_incremental_render(
2500 &mut cache,
2501 &mut rendered,
2502 &source,
2503 80,
2504 palette::PaletteMode::Dark,
2505 false,
2506 );
2507
2508 source.push_str("tail 世界\n");
2509 update_incremental_render(
2510 &mut cache,
2511 &mut rendered,
2512 &source,
2513 80,
2514 palette::PaletteMode::Dark,
2515 true,
2516 );
2517
2518 source.replace_range(..5, "ALPHA");
2519 update_incremental_render(
2520 &mut cache,
2521 &mut rendered,
2522 &source,
2523 80,
2524 palette::PaletteMode::Dark,
2525 false,
2526 );
2527 let mutated = render_markdown_tagged_with_palette(
2528 &source,
2529 80,
2530 Style::default(),
2531 palette::PaletteMode::Dark,
2532 );
2533 assert_eq!(
2534 rendered_fingerprint(&rendered),
2535 rendered_fingerprint(&mutated)
2536 );
2537
2538 update_incremental_render(
2539 &mut cache,
2540 &mut rendered,
2541 &source,
2542 37,
2543 palette::PaletteMode::Dark,
2544 true,
2545 );
2546 update_incremental_render(
2547 &mut cache,
2548 &mut rendered,
2549 &source,
2550 37,
2551 palette::PaletteMode::Light,
2552 true,
2553 );
2554
2555 let changed_style = Style::default().add_modifier(Modifier::ITALIC);
2556 update_incremental_render_with_style(
2557 &mut cache,
2558 &mut rendered,
2559 &source,
2560 37,
2561 changed_style,
2562 palette::PaletteMode::Light,
2563 true,
2564 );
2565 let rethemed = render_markdown_tagged_with_palette(
2566 &source,
2567 37,
2568 changed_style,
2569 palette::PaletteMode::Light,
2570 );
2571 assert_eq!(
2572 rendered_fingerprint(&rendered),
2573 rendered_fingerprint(&rethemed)
2574 );
2575 assert_eq!(cache.work().invalidations, 5);
2576 }
2577
2578 #[test]
2579 fn incremental_render_drops_committed_source_without_a_large_answer_cliff() {
2580 let line = format!("{}\n", "x".repeat(16 * 1024));
2581 let mut source = String::new();
2582 let mut cache = IncrementalMarkdownRenderCache::default();
2583 let mut rendered = Vec::new();
2584
2585 for index in 0..80 {
2586 source.push_str(&line);
2587 update_incremental_render(
2588 &mut cache,
2589 &mut rendered,
2590 &source,
2591 u16::MAX,
2592 palette::PaletteMode::Dark,
2593 index > 0,
2594 );
2595 assert_eq!(cache.retained_source_bytes(), 0);
2596 }
2597
2598 assert!(source.len() > 1024 * 1024);
2599 assert_eq!(cache.retained_source_bytes(), 0);
2600 assert_eq!(cache.work().classified_lines, 80);
2601 assert_eq!(cache.work().stable_blocks_rendered, 80);
2602 assert_eq!(cache.work().invalidations, 1);
2603 }
2604
2605 #[test]
2606 fn verified_append_resume_rejects_a_rewritten_committed_prefix() {
2607 // #6196: the streaming cache consumes the latex-rendered projection
2608 // of the raw stream. While a `\[` display block is open,
2609 // `render_latex_in_text` passes it through as literal text and the
2610 // incremental cache commits those lines. When the closing `\]`
2611 // finally arrives the rendered form rewrites the *committed* bytes
2612 // even though the raw stream only appended, so the append receipt
2613 // stays valid and only a content-aware committed-prefix check can
2614 // see the rewrite. A length-only check resumed from the stale lines
2615 // and the un-rendered opener stuck in the cell forever. The strings
2616 // below are the projections the latex layer produces for that
2617 // sequence.
2618 let mut cache = IncrementalMarkdownRenderCache::default();
2619 let mut rendered = Vec::new();
2620 // Beat 1: raw stream `para\n\[ \alpha\n` — the open block passes
2621 // through literally and its complete line is committed.
2622 update_incremental_render(
2623 &mut cache,
2624 &mut rendered,
2625 "para\n\\[ \\alpha\n",
2626 80,
2627 palette::PaletteMode::Dark,
2628 false,
2629 );
2630 assert_eq!(cache.work().invalidations, 1);
2631
2632 // Beat 2: the raw stream appended `\]` plus more prose; the
2633 // projection rewrites the committed line and grows past its old
2634 // length, so the resume guard cannot rely on length alone.
2635 let closed = "para\nα\nthe identity holds for every pair of terms\n";
2636 update_incremental_render(
2637 &mut cache,
2638 &mut rendered,
2639 closed,
2640 80,
2641 palette::PaletteMode::Dark,
2642 true,
2643 );
2644 assert_eq!(
2645 cache.work().invalidations,
2646 2,
2647 "rewriting committed bytes must invalidate the resume"
2648 );
2649 let cold = render_markdown_tagged_with_palette(
2650 closed,
2651 80,
2652 Style::default(),
2653 palette::PaletteMode::Dark,
2654 );
2655 assert_eq!(
2656 rendered_fingerprint(&rendered),
2657 rendered_fingerprint(&cold),
2658 "the re-render must match a cold render of the closed form"
2659 );
2660 assert!(
2661 !rendered.iter().any(|line| {
2662 line.line
2663 .spans
2664 .iter()
2665 .any(|span| span.content.contains("\\["))
2666 }),
2667 "the stale literal latex opener must not survive the close"
2668 );
2669 }
2670
2671 #[test]
2672 fn underscores_inside_identifiers_render_as_literal_text() {
2673 // Regression for PR #1455 / @tiger-dog: previously the inline
2674 // markdown parser ate the underscore in `codewhale_tui` because
2675 // it matched the `_italic_` pattern without a CommonMark-style
2676 // boundary check. The closing `_` followed by `t` (a letter)
2677 // must now be treated as part of the identifier, not as
2678 // markup. The same rule applies to `*` so identifiers like
2679 // `crate*foo` round-trip cleanly.
2680 let cases = [
2681 "crate codewhale_tui handles approvals",
2682 "see foo_bar_baz for details",
2683 "look at *not_emphasised*tail",
2684 ];
2685 for source in cases {
2686 let parsed = parse(source);
2687 let rendered: String = render_parsed(&parsed, 80, Style::default())
2688 .iter()
2689 .flat_map(|line| line.spans.iter().map(|span| span.content.as_ref()))
2690 .collect();
2691 // The original identifier (with underscores intact) must
2692 // appear in the rendered output. We don't assert on style
2693 // here — that's an implementation detail; we assert on
2694 // the user-visible character sequence.
2695 for token in source.split_whitespace().filter(|t| t.contains('_')) {
2696 assert!(
2697 rendered.contains(token),
2698 "identifier {token:?} must survive markdown rendering of {source:?}; got {rendered:?}"
2699 );
2700 }
2701 }
2702 }
2703
2704 #[test]
2705 fn underscore_emphasis_does_not_open_mid_word_across_math_subscripts() {
2706 // #6042: the closing-side guard alone let `b_p` open emphasis and
2707 // `t_?` close it, italicizing 60 characters of prose. CommonMark
2708 // forbids an intraword `_` from opening emphasis.
2709 let source = "[t_, b_p]. Actually — hold on, do we even tile all the way from t_?";
2710 let lines = render_parsed(&parse(source), 200, Style::default());
2711 let italic: Vec<&str> = lines
2712 .iter()
2713 .flat_map(|line| line.spans.iter())
2714 .filter(|span| span.style.add_modifier.contains(Modifier::ITALIC))
2715 .map(|span| span.content.as_ref())
2716 .collect();
2717 assert!(
2718 italic.is_empty(),
2719 "math subscripts must not open emphasis; italic spans: {italic:?}"
2720 );
2721
2722 // A properly flanked `_italic_` run still renders italic…
2723 let flanked = render_parsed(&parse("an _emphasised_ word"), 80, Style::default());
2724 assert!(
2725 flanked
2726 .iter()
2727 .flat_map(|line| line.spans.iter())
2728 .any(|span| span.style.add_modifier.contains(Modifier::ITALIC)
2729 && span.content.as_ref() == "emphasised"),
2730 "a flanked _italic_ run must still render italic"
2731 );
2732 // …and after punctuation it still opens.
2733 let after_punct = render_parsed(&parse("word (_also_)"), 80, Style::default());
2734 assert!(
2735 after_punct
2736 .iter()
2737 .flat_map(|line| line.spans.iter())
2738 .any(|span| span.style.add_modifier.contains(Modifier::ITALIC)
2739 && span.content.as_ref() == "also"),
2740 "a _ run after punctuation must still open"
2741 );
2742 }
2743
2744 #[test]
2745 fn render_markdown_matches_parse_then_render() {
2746 let source = "# Title\n\nA paragraph with a https://example.com link.\n\n- one\n- two\n```\ncode\n```";
2747 let direct = render_markdown(source, 80, Style::default())
2748 .iter()
2749 .flat_map(|l| l.spans.iter().map(|s| s.content.as_ref()))
2750 .collect::<String>();
2751 let parsed = parse(source);
2752 let two_step = render_parsed(&parsed, 80, Style::default())
2753 .iter()
2754 .flat_map(|l| l.spans.iter().map(|s| s.content.as_ref()))
2755 .collect::<String>();
2756 assert_eq!(direct, two_step);
2757 }
2758
2759 #[test]
2760 fn render_plain_text_preserves_literal_markdown_and_spacing() {
2761 let source = " # heading\n- item\n \nhello world\n";
2762 let lines = render_plain_text(source, 80, Style::default());
2763
2764 assert_eq!(
2765 visible_lines(&lines),
2766 vec![" # heading", "- item", " ", "hello world", ""]
2767 );
2768 }
2769
2770 #[test]
2771 fn render_plain_text_wraps_without_collapsing_spaces() {
2772 let source = "alpha beta gamma";
2773 let lines = render_plain_text(source, 12, Style::default());
2774 for width in rendered_widths(&lines) {
2775 assert!(width <= 12, "rendered width {width} exceeds budget");
2776 }
2777
2778 let combined = visible_lines(&lines).join("");
2779 assert_eq!(combined, source);
2780 }
2781
2782 #[test]
2783 fn render_plain_text_breaks_overlong_words() {
2784 let source = "x".repeat(40);
2785 let lines = render_plain_text(&source, 9, Style::default());
2786 for width in rendered_widths(&lines) {
2787 assert!(width <= 9, "rendered width {width} exceeds budget");
2788 }
2789
2790 let combined = visible_lines(&lines).join("");
2791 assert_eq!(combined, source);
2792 }
2793
2794 #[test]
2795 fn parse_is_width_independent() {
2796 // Same source, two parses, must produce identical AST. (Sanity:
2797 // parse must not depend on hidden global state like terminal width.)
2798 let source = "Hello\n\n## Heading\n- list\n";
2799 let a = parse(source);
2800 let b = parse(source);
2801 assert_eq!(a, b);
2802 }
2803
2804 #[test]
2805 fn render_parsed_word_wrap_changes_with_width() {
2806 // The same AST must produce different layouts at different widths;
2807 // otherwise the split is decorative, not functional.
2808 let parsed = parse("alpha beta gamma delta epsilon zeta");
2809 let wide = render_parsed(&parsed, 80, Style::default());
2810 let narrow = render_parsed(&parsed, 10, Style::default());
2811 assert!(
2812 narrow.len() > wide.len(),
2813 "narrow should produce more lines"
2814 );
2815 }
2816
2817 #[test]
2818 fn parse_invocations_increment() {
2819 // Counter is thread-local, so concurrent tests calling `parse()`
2820 // can't pollute each other.
2821 reset_parse_invocation_count();
2822 let _ = parse("hello\n");
2823 let _ = parse("world\n");
2824 assert_eq!(parse_invocation_count(), 2);
2825 }
2826
2827 #[test]
2828 fn render_parsed_does_not_call_parse() {
2829 // Width-only changes must hit only the render path. This is the
2830 // perf invariant CX#6 was filed for.
2831 let parsed = parse("multiline\nsource\nwith several\nlines\n");
2832 reset_parse_invocation_count();
2833 let _ = render_parsed(&parsed, 80, Style::default());
2834 let _ = render_parsed(&parsed, 40, Style::default());
2835 let _ = render_parsed(&parsed, 20, Style::default());
2836 assert_eq!(
2837 parse_invocation_count(),
2838 0,
2839 "render_parsed must not call parse"
2840 );
2841 }
2842
2843 // -----------------------------------------------------------------
2844 // #3897 — streaming re-parse is incremental, not quadratic
2845 // -----------------------------------------------------------------
2846
2847 /// Markdown corpus chosen to hit every carry-state transition the parser
2848 /// has: fences opened and closed across chunk boundaries, an unterminated
2849 /// fence, headings, nested lists, tables, rules, blanks, CRLF, and CJK.
2850 fn streaming_corpus() -> Vec<&'static str> {
2851 vec![
2852 "# Title\n\nSome prose that wraps.\n\n- alpha\n- beta\n",
2853 "text\n```rust\nlet x = 1;\nlet y = 2;\n```\nafter\n",
2854 "```\nunterminated fence never closes\nstill inside\n",
2855 "| a | b |\n|---|---|\n| 1 | 2 |\n\n---\n\ndone\n",
2856 "1. one\n2. two\n * nested\n\n## Sub\n\n***\n",
2857 "混合 CJK 内容\n\n```python\nprint(\"中文\")\n```\n尾部\n",
2858 "crlf lines\r\nsecond\r\n\r\n```go\nfmt.Println()\r\n```\r\n",
2859 "no trailing newline at all",
2860 "",
2861 ]
2862 }
2863
2864 /// The acceptance guarantee: streaming a message chunk by chunk produces,
2865 /// at every intermediate prefix, exactly what a full re-parse of that
2866 /// prefix produces. Byte-for-byte on the AST, so a divergence in any field
2867 /// of any block fails here rather than showing up as a render artifact.
2868 #[test]
2869 fn incremental_parse_matches_a_full_reparse_at_every_prefix() {
2870 for source in streaming_corpus() {
2871 // Grow one byte at a time (respecting char boundaries) — the
2872 // worst case for a parser that commits too eagerly.
2873 for end in 0..=source.len() {
2874 if !source.is_char_boundary(end) {
2875 continue;
2876 }
2877 let prefix = &source[..end];
2878 let streamed = parse(prefix);
2879
2880 // A cold parser is the reference: no memo, no resumption.
2881 let mut cold = ParseState::default();
2882 cold.commit_complete_lines(prefix);
2883 let reference = cold.snapshot(prefix);
2884
2885 assert_eq!(
2886 streamed, reference,
2887 "prefix {end} of {source:?} diverged from a full re-parse"
2888 );
2889 }
2890 }
2891 }
2892
2893 /// The performance guarantee: work per chunk must not grow with the
2894 /// message. Counted in lines actually classified, which is the quantity
2895 /// that was quadratic — the old code re-classified every line on every
2896 /// chunk.
2897 #[test]
2898 fn streaming_does_not_reclassify_committed_lines() {
2899 let chunk = "a line of prose\n";
2900 let chunks = 400;
2901
2902 let mut content = String::new();
2903 let mut state = ParseState::default();
2904 let mut total_committed = 0usize;
2905
2906 for _ in 0..chunks {
2907 content.push_str(chunk);
2908 assert!(
2909 state.can_resume_from(&content),
2910 "an append-only stream must always be resumable"
2911 );
2912 let before = state.blocks.len();
2913 state.commit_complete_lines(&content);
2914 total_committed += state.blocks.len() - before;
2915 }
2916
2917 // Quadratic would be chunks * (chunks + 1) / 2 = 80,200 classifications.
2918 assert_eq!(
2919 total_committed, chunks,
2920 "each line must be classified exactly once across the whole stream"
2921 );
2922 assert_eq!(state.blocks.len(), chunks);
2923 }
2924
2925 /// Resumption is verified, never assumed. Content that does not extend the
2926 /// committed prefix must fall back to a full parse rather than splice
2927 /// unrelated blocks together.
2928 #[test]
2929 fn a_changed_prefix_is_not_resumable() {
2930 let mut state = ParseState::default();
2931 state.commit_complete_lines("first line\nsecond line\n");
2932
2933 assert!(state.can_resume_from("first line\nsecond line\nthird\n"));
2934 // Earlier bytes rewritten.
2935 assert!(!state.can_resume_from("FIRST line\nsecond line\nthird\n"));
2936 // Buffer shrank (a different, shorter cell).
2937 assert!(!state.can_resume_from("first line\n"));
2938 // Entirely unrelated content.
2939 assert!(!state.can_resume_from("something else\n"));
2940 }
2941
2942 /// Interleaving two different sources through the shared memo must not
2943 /// contaminate either — this is the multi-cell render-loop case.
2944 #[test]
2945 fn interleaved_sources_do_not_contaminate_each_other() {
2946 let a = "# Alpha\n\nalpha body\n";
2947 let b = "```rust\nlet b = 1;\n```\n";
2948 for _ in 0..5 {
2949 assert_eq!(parse(a), reference_parse(a));
2950 assert_eq!(parse(b), reference_parse(b));
2951 }
2952 }
2953
2954 fn reference_parse(content: &str) -> ParsedMarkdown {
2955 let mut cold = ParseState::default();
2956 cold.commit_complete_lines(content);
2957 cold.snapshot(content)
2958 }
2959
2960 #[test]
2961 fn fenced_code_block_collected_in_parse() {
2962 let parsed = parse("text\n```rust\ncode line one\ncode line two\n```\nmore\n");
2963 let blocks = &parsed.blocks;
2964 // text paragraph, two code lines, more paragraph (fences are dropped)
2965 let code_lines: Vec<_> = blocks
2966 .iter()
2967 .filter_map(|b| match b {
2968 Block::Code {
2969 line,
2970 language,
2971 block_id,
2972 } => Some((line.as_str(), language.as_deref(), *block_id)),
2973 _ => None,
2974 })
2975 .collect();
2976 assert_eq!(
2977 code_lines,
2978 vec![
2979 ("code line one", Some("rust"), 1),
2980 ("code line two", Some("rust"), 1),
2981 ]
2982 );
2983 }
2984
2985 #[test]
2986 fn adjacent_code_fences_keep_distinct_highlighter_state() {
2987 let parsed = parse("```rust\n/* open\n```\n```rust\nlet x = 1;\n```\n");
2988 let ids = parsed
2989 .blocks
2990 .iter()
2991 .filter_map(|block| match block {
2992 Block::Code { block_id, .. } => Some(*block_id),
2993 _ => None,
2994 })
2995 .collect::<Vec<_>>();
2996 assert_eq!(ids, vec![1, 2]);
2997 }
2998
2999 #[test]
3000 fn rust_fence_respects_terminal_color_depth_without_reserved_rgb() {
3001 let rendered = render_markdown_tagged(
3002 "```rust\nfn main() {\n let answer: u32 = 42; // comment\n}\n```",
3003 100,
3004 Style::default(),
3005 );
3006 let colors = rendered
3007 .iter()
3008 .flat_map(|line| line.line.spans.iter())
3009 .filter_map(|span| span.style.fg)
3010 .collect::<std::collections::HashSet<_>>();
3011 if syntax_color_depth() == palette::ColorDepth::Monochrome {
3012 assert_eq!(colors, [Color::Reset].into(), "NO_COLOR: {colors:?}");
3013 } else {
3014 assert!(colors.len() > 1, "expected syntax colors, got: {colors:?}");
3015 }
3016 for color in colors {
3017 assert_ne!(color, palette::WHALE_HUMAN);
3018 assert_ne!(color, palette::WHALE_LIVE);
3019 assert_ne!(color, palette::WHALE_ACTION);
3020 assert_ne!(color, palette::WHALE_ERROR);
3021 }
3022 }
3023
3024 #[test]
3025 fn syntax_colors_use_existing_depth_quantizer_and_grayscale_path() {
3026 assert!(matches!(
3027 syntax_rgb_to_terminal_color(
3028 120,
3029 80,
3030 200,
3031 palette::PaletteMode::Dark,
3032 palette::ColorDepth::Ansi256,
3033 ),
3034 Color::Indexed(_)
3035 ));
3036 assert!(matches!(
3037 syntax_rgb_to_terminal_color(
3038 120,
3039 80,
3040 200,
3041 palette::PaletteMode::Dark,
3042 palette::ColorDepth::Ansi16,
3043 ),
3044 Color::Black
3045 | Color::Red
3046 | Color::Green
3047 | Color::Yellow
3048 | Color::Blue
3049 | Color::Magenta
3050 | Color::Cyan
3051 | Color::Gray
3052 | Color::DarkGray
3053 | Color::LightRed
3054 | Color::LightGreen
3055 | Color::LightYellow
3056 | Color::LightBlue
3057 | Color::LightMagenta
3058 | Color::LightCyan
3059 | Color::White
3060 ));
3061 let gray = syntax_rgb_to_terminal_color(
3062 120,
3063 80,
3064 200,
3065 palette::PaletteMode::Grayscale,
3066 palette::ColorDepth::TrueColor,
3067 );
3068 assert!(matches!(gray, Color::Rgb(r, g, b) if r == g && g == b));
3069 }
3070
3071 #[test]
3072 fn syntax_assets_are_lazy_singletons_and_explicit_modes_select_themes() {
3073 assert!(std::ptr::eq(syntax_set(), syntax_set()));
3074 assert!(std::ptr::eq(theme_set(), theme_set()));
3075 assert!(!std::ptr::eq(
3076 selected_syntax_theme(palette::PaletteMode::Dark),
3077 selected_syntax_theme(palette::PaletteMode::Light),
3078 ));
3079 }
3080
3081 #[test]
3082 fn depth_quantization_cannot_reintroduce_reserved_semantic_colors() {
3083 let reserved = [
3084 palette::WHALE_HUMAN,
3085 palette::WHALE_LIVE,
3086 palette::WHALE_ACTION,
3087 palette::WHALE_ERROR,
3088 ];
3089 for depth in [
3090 palette::ColorDepth::TrueColor,
3091 palette::ColorDepth::Ansi256,
3092 palette::ColorDepth::Ansi16,
3093 ] {
3094 let reserved_at_depth = reserved.map(|color| palette::adapt_color(color, depth));
3095 for semantic in reserved {
3096 let Color::Rgb(r, g, b) = semantic else {
3097 panic!("reserved syntax guard expects RGB semantic colors");
3098 };
3099 let syntax =
3100 syntax_rgb_to_terminal_color(r, g, b, palette::PaletteMode::Dark, depth);
3101 assert!(
3102 !reserved_at_depth.contains(&syntax),
3103 "{syntax:?} reintroduced a reserved color at {depth:?}"
3104 );
3105 }
3106 }
3107 }
3108
3109 #[test]
3110 fn code_block_indentation_is_preserved_in_render() {
3111 // Leading whitespace in code blocks is semantic — indented lines must
3112 // not be stripped to column zero when rendered.
3113 let md = "```\nfn main() {\n println!(\"hi\");\n}\n```\n";
3114 let lines = render_markdown(md, 80, Style::default());
3115 let text: Vec<String> = lines
3116 .iter()
3117 .map(|l| {
3118 l.spans
3119 .iter()
3120 .map(|s| s.content.as_ref())
3121 .collect::<String>()
3122 })
3123 .collect();
3124 // The indented line must start with spaces (the 2-space code prefix
3125 // plus the 4-space source indentation).
3126 let indented = text
3127 .iter()
3128 .find(|t| t.contains("println"))
3129 .expect("should find println line");
3130 assert!(
3131 indented.starts_with(" "),
3132 "expected 6+ leading spaces (2 block prefix + 4 indent), got: {indented:?}"
3133 );
3134 }
3135
3136 #[test]
3137 fn wrap_code_line_preserves_leading_whitespace() {
3138 // A short line must not be modified.
3139 assert_eq!(wrap_code_line(" let x = 1;", 80), vec![" let x = 1;"]);
3140
3141 // A line that exceeds the width must be hard-wrapped, keeping the
3142 // leading whitespace on the first chunk.
3143 let chunks = wrap_code_line(" abcdefgh", 8);
3144 assert_eq!(chunks[0], " abcd", "first chunk keeps leading spaces");
3145 assert_eq!(chunks[1], "efgh");
3146
3147 // Empty line produces one empty chunk.
3148 assert_eq!(wrap_code_line("", 80), vec![""]);
3149 }
3150
3151 #[test]
3152 fn wrap_code_line_tab_counts_toward_width() {
3153 // tab (8 cols) + "xy" (2 cols) = 10 ≤ 10 — fits on one line.
3154 let chunks = wrap_code_line("\txy", 10);
3155 assert_eq!(chunks, vec!["\txy"], "tab + 2 chars fits in width 10");
3156
3157 // tab (8 cols) + "x" (1 col) = 9 ≤ 9 — "x" fits; "y" overflows.
3158 let chunks = wrap_code_line("\txy", 9);
3159 assert_eq!(chunks[0], "\tx", "tab + first char fits exactly");
3160 assert_eq!(chunks[1], "y", "second char wraps");
3161
3162 // tab alone (8 cols) fits in width 8; the next "x" overflows.
3163 let chunks = wrap_code_line("\tx", 8);
3164 assert_eq!(chunks[0], "\t");
3165 assert_eq!(chunks[1], "x");
3166 }
3167
3168 #[test]
3169 fn markdown_grapheme_width_uses_tab_stop_and_string_width() {
3170 // At column 0 a tab fills to column 8.
3171 assert_eq!(markdown_grapheme_width("\t", 0), 8);
3172 // At column 4 a tab fills to column 8 (4 remaining).
3173 assert_eq!(markdown_grapheme_width("\t", 4), 4);
3174 // At column 8 a tab fills to the next stop at 16 (8 columns).
3175 assert_eq!(markdown_grapheme_width("\t", 8), 8);
3176 // Regular ASCII is 1.
3177 assert_eq!(markdown_grapheme_width("a", 0), 1);
3178 // A fully-qualified keycap is one two-column grapheme.
3179 assert_eq!(markdown_grapheme_width("1\u{fe0f}\u{20e3}", 0), 2);
3180 }
3181
3182 #[test]
3183 fn ordered_and_unordered_list_items_parse() {
3184 let parsed = parse("- alpha\n* beta\n1. gamma\n");
3185 let items: Vec<_> = parsed
3186 .blocks
3187 .iter()
3188 .filter_map(|b| match b {
3189 Block::ListItem { bullet, text } => Some((bullet.as_str(), text.as_str())),
3190 _ => None,
3191 })
3192 .collect();
3193 assert_eq!(items, vec![("-", "alpha"), ("-", "beta"), ("1.", "gamma")]);
3194 }
3195
3196 #[test]
3197 fn blockquote_lines_parse_with_depth() {
3198 let parsed =
3199 parse("> hello\n>\n>> nested\n> > spaced\n>no-space\n>\t tabbed\nlone > arrow\n");
3200 let quotes: Vec<_> = parsed
3201 .blocks
3202 .iter()
3203 .filter_map(|b| match b {
3204 Block::Quote { depth, text } => Some((*depth, text.as_str())),
3205 _ => None,
3206 })
3207 .collect();
3208 assert_eq!(
3209 quotes,
3210 vec![
3211 (1, "hello"),
3212 (1, ""),
3213 (2, "nested"),
3214 (2, "spaced"),
3215 (1, "no-space"),
3216 (1, "tabbed"),
3217 ]
3218 );
3219 // `lone > arrow` starts with prose, so it stays a paragraph.
3220 assert_eq!(parsed.blocks.len(), 7);
3221 }
3222
3223 #[test]
3224 fn code_fence_contains_quote_lines_untouched() {
3225 // Fenced-code lines are collected before blockquote classification, so
3226 // `>` inside a fence must stay literal code content.
3227 let parsed = parse("```\n> not a quote\n\n> but this is\n```\n");
3228 let code: Vec<_> = parsed
3229 .blocks
3230 .iter()
3231 .filter_map(|b| match b {
3232 Block::Code { line, .. } => Some(line.as_str()),
3233 _ => None,
3234 })
3235 .collect();
3236 assert_eq!(code, vec!["> not a quote", "", "> but this is"]);
3237 assert!(
3238 parsed
3239 .blocks
3240 .iter()
3241 .all(|b| !matches!(b, Block::Quote { .. })),
3242 "lines inside a fence must stay code, never quotes"
3243 );
3244 }
3245
3246 #[test]
3247 fn four_backtick_fence_keeps_shorter_fence_and_quotes_as_code() {
3248 // A ```` opener must not be closed by a shorter ``` line: the shorter
3249 // fence and any `>` lines after it stay code content (CommonMark
3250 // fence-length rule), and the block only closes on a fence >= opener.
3251 let parsed = parse("````\n```\n> still code\n`````\n> now a quote\n");
3252 let code: Vec<_> = parsed
3253 .blocks
3254 .iter()
3255 .filter_map(|b| match b {
3256 Block::Code { line, .. } => Some(line.as_str()),
3257 _ => None,
3258 })
3259 .collect();
3260 assert_eq!(code, vec!["```", "> still code"]);
3261 let quotes: Vec<_> = parsed
3262 .blocks
3263 .iter()
3264 .filter_map(|b| match b {
3265 Block::Quote { text, .. } => Some(text.as_str()),
3266 _ => None,
3267 })
3268 .collect();
3269 assert_eq!(quotes, vec!["now a quote"]);
3270 }
3271
3272 #[test]
3273 fn longer_fence_closes_shorter_opener() {
3274 // CommonMark: a closing fence may be longer than the opener; only the
3275 // opener's length is the minimum.
3276 let parsed = parse("```\ncode\n````\nplain\n");
3277 let code: Vec<_> = parsed
3278 .blocks
3279 .iter()
3280 .filter_map(|b| match b {
3281 Block::Code { line, .. } => Some(line.as_str()),
3282 _ => None,
3283 })
3284 .collect();
3285 assert_eq!(code, vec!["code"]);
3286 let paragraphs: Vec<_> = parsed
3287 .blocks
3288 .iter()
3289 .filter_map(|b| match b {
3290 Block::Paragraph { text } => Some(text.as_str()),
3291 _ => None,
3292 })
3293 .collect();
3294 assert_eq!(paragraphs, vec!["plain"]);
3295 }
3296
3297 #[test]
3298 fn backticks_with_info_do_not_close_an_open_fence() {
3299 let parsed = parse("```\n```rust\n> still code\n```\n");
3300 let code: Vec<_> = parsed
3301 .blocks
3302 .iter()
3303 .filter_map(|block| match block {
3304 Block::Code { line, .. } => Some(line.as_str()),
3305 _ => None,
3306 })
3307 .collect();
3308
3309 assert_eq!(code, vec!["```rust", "> still code"]);
3310 assert!(
3311 parsed
3312 .blocks
3313 .iter()
3314 .all(|block| !matches!(block, Block::Quote { .. }))
3315 );
3316 }
3317
3318 #[test]
3319 fn blockquote_renders_rail_and_inline_formatting() {
3320 let rendered = render_markdown_tagged(
3321 "> **bold** `code` and see https://example.com",
3322 80,
3323 Style::default(),
3324 );
3325 assert_eq!(
3326 tagged_visible(&rendered),
3327 vec!["│ bold code and see https://example.com"]
3328 );
3329 let spans = &rendered[0].line.spans;
3330 assert_eq!(spans[0].content, "│ ");
3331 assert!(
3332 spans
3333 .iter()
3334 .any(|span| span.content == "bold"
3335 && span.style.add_modifier.contains(Modifier::BOLD)),
3336 "inline bold must survive inside a quote"
3337 );
3338 assert!(
3339 spans
3340 .iter()
3341 .any(|span| span.content == "code"
3342 && span.style.bg == Some(palette::SURFACE_ELEVATED)),
3343 "inline code is styled distinctly from plain text"
3344 );
3345 // First-row rail is stripped by the transcript rail-scan along with
3346 // the assistant glyph, so copy must report 0 here (mirror list items)
3347 // — reporting the rail width would strip it twice and lose text.
3348 assert_eq!(rendered[0].copy_prefix_width, 0);
3349 assert_eq!(
3350 rendered[0].links,
3351 vec![osc8::LineLink {
3352 col_start: 20,
3353 col_end: 38,
3354 target: "https://example.com".to_string(),
3355 }]
3356 );
3357 }
3358
3359 #[test]
3360 fn nested_blockquote_renders_multiple_rails_capped() {
3361 let rendered =
3362 render_markdown_tagged(">>> deep\n>>>>>>>>>> too deep\n", 80, Style::default());
3363 assert_eq!(
3364 tagged_visible(&rendered),
3365 vec!["│ │ │ deep", "│ │ │ │ too deep"]
3366 );
3367 assert_eq!(
3368 rendered[0]
3369 .line
3370 .spans
3371 .iter()
3372 .take(3)
3373 .map(|span| span.content.as_ref())
3374 .collect::<Vec<_>>(),
3375 vec!["│ ", "│ ", "│ "],
3376 "each rail must remain independently discoverable by selection copy"
3377 );
3378 }
3379
3380 #[test]
3381 fn blockquote_wraps_with_continuation_rail_indent() {
3382 let source = "> alpha beta gamma delta epsilon zeta";
3383 let rendered = render_markdown_tagged(source, 12, Style::default());
3384 let visible = tagged_visible(&rendered);
3385 assert!(visible.len() > 1, "fixture must wrap: {visible:?}");
3386 assert!(visible[0].starts_with("│ "), "first row starts with rail");
3387 // Copy-prefix accounting: first row reports 0 (rail-scan strips it),
3388 // continuation rows report the rail width (plain spaces are not
3389 // stripped by the rail-scan).
3390 assert_eq!(rendered[0].copy_prefix_width, 0);
3391 for row in rendered.iter().skip(1) {
3392 assert_eq!(row.copy_prefix_width, 2);
3393 }
3394 for row in &visible[1..] {
3395 assert!(
3396 row.starts_with(" "),
3397 "continuation rows keep the rail width indent: {row:?}"
3398 );
3399 assert!(
3400 !row.starts_with('│'),
3401 "rail appears only on the first row: {row:?}"
3402 );
3403 }
3404 for width in rendered.iter().map(|row| {
3405 row.line
3406 .spans
3407 .iter()
3408 .map(|span| span.content.as_ref().width())
3409 .sum::<usize>()
3410 }) {
3411 assert!(width <= 12, "rendered width {width} exceeds budget");
3412 }
3413 // Rows re-join into the quote content (rail and alignment stripped).
3414 let combined = visible
3415 .iter()
3416 .map(|row| row.trim_start_matches('│').trim_start())
3417 .collect::<Vec<_>>()
3418 .join(" ");
3419 assert_eq!(combined, &source[2..]);
3420 }
3421
3422 fn tagged_visible(lines: &[RenderedMarkdownLine]) -> Vec<String> {
3423 lines
3424 .iter()
3425 .map(|rendered| {
3426 rendered
3427 .line
3428 .spans
3429 .iter()
3430 .map(|span| span.content.as_ref())
3431 .collect()
3432 })
3433 .collect()
3434 }
3435
3436 #[test]
3437 fn http_links_keep_visible_text_and_out_of_band_metadata() {
3438 let source = "see https://example.com for details";
3439 let rendered = render_markdown_tagged(source, 80, Style::default());
3440 assert_eq!(tagged_visible(&rendered), vec![source]);
3441 assert!(
3442 rendered
3443 .iter()
3444 .flat_map(|line| &line.line.spans)
3445 .all(|span| { !span.content.contains('\x1b') && !span.content.contains("]8;;") }),
3446 "escape payloads must never enter visible spans"
3447 );
3448 assert_eq!(
3449 rendered[0].links,
3450 vec![osc8::LineLink {
3451 col_start: 4,
3452 col_end: 22,
3453 target: "https://example.com".to_string(),
3454 }]
3455 );
3456 }
3457
3458 #[test]
3459 fn bare_http_links_exclude_surrounding_punctuation_from_target() {
3460 let source = "see (https://example.com/path).";
3461 let rendered = render_markdown_tagged(source, 80, Style::default());
3462 assert_eq!(tagged_visible(&rendered), vec![source]);
3463 assert_eq!(rendered[0].links.len(), 1);
3464 let link = &rendered[0].links[0];
3465 assert_eq!(link.target, "https://example.com/path");
3466 assert_eq!(link.col_start, 5);
3467 assert_eq!(link.col_end, 28);
3468 }
3469
3470 #[test]
3471 fn bare_http_links_preserve_balanced_parentheses_in_target() {
3472 let url = "https://en.wikipedia.org/wiki/Function_(mathematics)";
3473 let source = format!("see {url}.");
3474 let rendered = render_markdown_tagged(&source, 100, Style::default());
3475 assert_eq!(tagged_visible(&rendered), vec![source]);
3476 assert_eq!(rendered[0].links.len(), 1);
3477 assert_eq!(rendered[0].links[0].target, url);
3478 }
3479
3480 #[test]
3481 fn wrapped_url_chunks_keep_visible_label_and_full_target() {
3482 let url = "https://raw.githubusercontent.com/Hmbown/deepseek-skills/main/index.json";
3483 let rendered = render_markdown_tagged(url, 34, Style::default());
3484 let visible = tagged_visible(&rendered);
3485 assert!(visible.len() > 1, "fixture must wrap: {visible:?}");
3486 assert_eq!(visible.concat(), url);
3487 for (line, text) in rendered.iter().zip(&visible) {
3488 assert_eq!(line.links.len(), 1, "each wrapped chunk is linked");
3489 assert_eq!(line.links[0].target, url);
3490 assert_eq!(line.links[0].col_start, 0);
3491 assert_eq!(line.links[0].col_end, text.width().saturating_sub(1));
3492 assert!(!text.contains('\x1b') && !text.contains("]8;;"));
3493 }
3494 }
3495
3496 #[test]
3497 fn named_link_shows_only_label_and_keeps_target_in_metadata() {
3498 let rendered = render_markdown_tagged(
3499 "read [the docs](https://example.com/guide) now",
3500 80,
3501 Style::default(),
3502 );
3503 assert_eq!(tagged_visible(&rendered), vec!["read the docs now"]);
3504 assert_eq!(
3505 rendered[0].links,
3506 vec![osc8::LineLink {
3507 col_start: 5,
3508 col_end: 12,
3509 target: "https://example.com/guide".to_string(),
3510 }]
3511 );
3512 }
3513
3514 #[test]
3515 fn named_links_reject_non_web_schemes_and_normalize_http_scheme() {
3516 let unsafe_link = render_markdown_tagged("[run](javascript:alert)", 80, Style::default());
3517 assert_eq!(tagged_visible(&unsafe_link), vec!["run"]);
3518 assert!(unsafe_link.iter().all(|line| line.links.is_empty()));
3519
3520 let web_link =
3521 render_markdown_tagged("[docs](HTTPS://example.com/guide)", 80, Style::default());
3522 assert_eq!(tagged_visible(&web_link), vec!["docs"]);
3523 assert_eq!(web_link[0].links[0].target, "https://example.com/guide");
3524 }
3525
3526 #[test]
3527 fn named_links_target_absolute_paths_with_the_file_scheme() {
3528 let rendered = render_markdown_tagged(
3529 "edit [main.rs](/repo/src/main.rs) now",
3530 80,
3531 Style::default(),
3532 );
3533 assert_eq!(tagged_visible(&rendered), vec!["edit main.rs now"]);
3534 assert_eq!(
3535 rendered[0].links,
3536 vec![osc8::LineLink {
3537 col_start: 5,
3538 col_end: 11,
3539 target: "file:///repo/src/main.rs".to_string(),
3540 }]
3541 );
3542
3543 let explicit =
3544 render_markdown_tagged("[main.rs](FILE:///repo/src/main.rs)", 80, Style::default());
3545 assert_eq!(explicit[0].links[0].target, "file:///repo/src/main.rs");
3546 }
3547
3548 #[test]
3549 fn named_links_reject_relative_paths_and_smuggled_control_bytes() {
3550 // Nothing here resolves a relative path against a workspace root, so a
3551 // relative destination would link to whatever the terminal's cwd is.
3552 let relative = render_markdown_tagged("[main.rs](src/main.rs)", 80, Style::default());
3553 assert_eq!(tagged_visible(&relative), vec!["main.rs"]);
3554 assert!(relative.iter().all(|line| line.links.is_empty()));
3555
3556 let hostile = render_markdown_tagged(
3557 "[log](/tmp/a\x07b\x1b]8;;https://evil.test\x1b\\c)",
3558 80,
3559 Style::default(),
3560 );
3561 assert!(
3562 hostile.iter().all(|line| line.links.is_empty()),
3563 "control bytes must not reach a link target: {hostile:?}"
3564 );
3565
3566 // A `file://host/share` destination is a remote reference, not a local
3567 // path, and must not be rewritten into one.
3568 let host = render_markdown_tagged("[share](file://evil.test/etc)", 80, Style::default());
3569 assert!(host.iter().all(|line| line.links.is_empty()));
3570 }
3571
3572 #[test]
3573 fn bare_paths_in_prose_are_never_linkified() {
3574 let rendered = render_markdown_tagged(
3575 "the fix landed in /repo/src/main.rs today",
3576 80,
3577 Style::default(),
3578 );
3579 assert!(rendered.iter().all(|line| line.links.is_empty()));
3580 }
3581
3582 #[test]
3583 fn table_separator_row_is_kept() {
3584 // Separator rows are now kept as TableSeparator blocks so the
3585 // renderer can draw horizontal rules at the correct positions.
3586 let src = "| 项目属性 | 详情 |\n|----------|------|\n| **语言** | Rust 1.88+ |\n";
3587 let parsed = parse(src);
3588 let blocks: Vec<_> = parsed.blocks.iter().collect();
3589 // Should have 2 TableRow blocks (header + data) + 1 TableSeparator
3590 let table_rows: Vec<_> = blocks
3591 .iter()
3592 .filter(|b| matches!(b, Block::TableRow(_)))
3593 .collect();
3594 assert_eq!(table_rows.len(), 2, "expected 2 table rows: {blocks:?}");
3595 let separators: Vec<_> = blocks
3596 .iter()
3597 .filter(|b| matches!(b, Block::TableSeparator))
3598 .collect();
3599 assert_eq!(
3600 separators.len(),
3601 1,
3602 "expected 1 table separator: {blocks:?}"
3603 );
3604 }
3605
3606 #[test]
3607 fn bold_markers_stripped_in_render() {
3608 let src = "这是一个 **Rust 工作区项目**,包含多个 crate。\n";
3609 let lines = render_markdown(src, 80, Style::default());
3610 let text: String = lines
3611 .iter()
3612 .flat_map(|l| l.spans.iter().map(|s| s.content.as_ref()))
3613 .collect();
3614 assert!(
3615 !text.contains("**"),
3616 "bold markers leaked into output: {text:?}"
3617 );
3618 assert!(text.contains("Rust"), "bold content missing: {text:?}");
3619 }
3620
3621 #[test]
3622 fn table_renders_with_box_drawing_borders() {
3623 let src = "| 文件 | 改动 |\n|---|---|\n| foo.rs | 重写 |\n";
3624 let lines = render_markdown(src, 60, Style::default());
3625 let text: String = lines
3626 .iter()
3627 .flat_map(|l| l.spans.iter().map(|s| s.content.as_ref()))
3628 .collect();
3629 // Column pipes still present
3630 assert!(text.contains('│'), "table pipe separator missing: {text:?}");
3631 // Separator row rendered as middle border, not raw markdown
3632 assert!(
3633 !text.contains("|---|"),
3634 "raw separator row leaked: {text:?}"
3635 );
3636 // Top and bottom borders present
3637 assert!(
3638 text.contains('\u{250C}'),
3639 "top-left corner missing: {text:?}"
3640 );
3641 assert!(
3642 text.contains('\u{2510}'),
3643 "top-right corner missing: {text:?}"
3644 );
3645 assert!(
3646 text.contains('\u{2514}'),
3647 "bottom-left corner missing: {text:?}"
3648 );
3649 assert!(
3650 text.contains('\u{2518}'),
3651 "bottom-right corner missing: {text:?}"
3652 );
3653 // Middle separator present (at the |---|---| position)
3654 assert!(
3655 text.contains('\u{251C}'),
3656 "middle-left junction missing: {text:?}"
3657 );
3658 assert!(
3659 text.contains('\u{2524}'),
3660 "middle-right junction missing: {text:?}"
3661 );
3662 }
3663
3664 #[test]
3665 fn table_pipes_inside_inline_code_stay_in_the_cell() {
3666 let src = "| Check | Result |\n\
3667 |---|---|\n\
3668 | `strings ~/.cargo/bin/codewhale-tui | grep -c \"legacy marker\"` | 0 matches |\n";
3669 let parsed = parse(src);
3670
3671 let rows: Vec<&Vec<String>> = parsed
3672 .blocks
3673 .iter()
3674 .filter_map(|block| match block {
3675 Block::TableRow(cells) => Some(cells),
3676 _ => None,
3677 })
3678 .collect();
3679
3680 assert_eq!(rows.len(), 2, "expected header + data row: {rows:?}");
3681 assert_eq!(
3682 rows[1],
3683 &vec![
3684 "`strings ~/.cargo/bin/codewhale-tui | grep -c \"legacy marker\"`".to_string(),
3685 "0 matches".to_string(),
3686 ]
3687 );
3688
3689 let rendered_lines = visible_lines(&render_markdown(src, 200, Style::default()));
3690 let rendered = rendered_lines.join("\n");
3691 assert!(
3692 rendered.contains("grep -c"),
3693 "inline-code command was lost: {rendered}"
3694 );
3695 let data_line = rendered_lines
3696 .iter()
3697 .find(|line| line.contains("strings ~/.cargo/bin/codewhale-tui"))
3698 .expect("data row should render");
3699 assert_eq!(
3700 data_line.matches('│').count(),
3701 3,
3702 "two-column table row should have left, middle, and right separators: {data_line:?}"
3703 );
3704 }
3705
3706 /// Cells longer than the per-column width must word-wrap to multiple
3707 /// lines instead of getting truncated with `…`. Truncation silently
3708 /// drops content the user can never see — particularly bad in narrow
3709 /// Windows terminals or with verbose English/Chinese instructional
3710 /// tables (the common LLM-output case).
3711 #[test]
3712 fn table_cell_wider_than_column_wraps_instead_of_truncating() {
3713 let src = "| Feature | How to verify |\n\
3714 |---|---|\n\
3715 | Workspace-local commands | Drop a .deepseek/commands/foo.md in any project, run deepseek from there, type /foo — should dispatch |\n";
3716 let lines = render_markdown(src, 80, Style::default());
3717 let combined: String = lines
3718 .iter()
3719 .flat_map(|l| l.spans.iter().map(|s| s.content.as_ref()))
3720 .collect();
3721
3722 assert!(
3723 !combined.contains('…'),
3724 "table cell was truncated with `…` instead of wrapping; got: {combined:?}"
3725 );
3726 assert!(
3727 combined.contains("type /foo"),
3728 "tail of long cell was lost; got: {combined:?}"
3729 );
3730 assert!(
3731 combined.contains("Workspace-local commands"),
3732 "short cell content lost; got: {combined:?}"
3733 );
3734 }
3735
3736 /// Wrapped table rows must keep column separators on every visual
3737 /// line so the columns remain visually aligned across all wrapped
3738 /// segments. A wrapped row's continuation lines should still show
3739 /// the `│` separator pipes at the same column positions.
3740 #[test]
3741 fn wrapped_table_row_preserves_column_separators() {
3742 let src = "| A | B |\n\
3743 |---|---|\n\
3744 | short | this is a very very long second cell that absolutely must wrap to a new visual line because it cannot fit in the column allocated to it at this terminal width |\n";
3745 let lines = render_markdown(src, 60, Style::default());
3746 let rendered: Vec<String> = lines
3747 .iter()
3748 .map(|l| {
3749 l.spans
3750 .iter()
3751 .map(|s| s.content.as_ref())
3752 .collect::<String>()
3753 })
3754 .collect();
3755
3756 // Every line in the rendered table — including wrapped continuation
3757 // lines — must show the pipe column separator. We identify table
3758 // body lines as ones that start with the row separator `│`.
3759 let body_lines: Vec<&String> = rendered.iter().filter(|s| s.starts_with('│')).collect();
3760
3761 assert!(
3762 body_lines.len() >= 3,
3763 "expected at least header + multi-line data row (3+ body lines), got {}: {:?}",
3764 body_lines.len(),
3765 body_lines
3766 );
3767
3768 for line in &body_lines {
3769 assert!(
3770 line.matches('│').count() >= 3,
3771 "every wrapped table line should have N+1 column separators \
3772 for N columns; got fewer in: {line:?}"
3773 );
3774 }
3775
3776 // All of the long cell's content must appear across the wrapped lines.
3777 let combined: String = rendered.join("\n");
3778 for fragment in ["this is a very very long", "must wrap", "terminal width"] {
3779 assert!(
3780 combined.contains(fragment),
3781 "fragment {fragment:?} missing from wrapped output:\n{combined}"
3782 );
3783 }
3784 }
3785
3786 // ─── Paragraph wrap regression suite (#1344, #1351) ────────────────────
3787 //
3788 // The bug: paragraph wrap (render_line_with_links) and code-block wrap
3789 // (wrap_text) are word-based. A single word wider than the available
3790 // width was placed alone on a line and silently overflowed the right
3791 // edge of the transcript. Long URLs / paths / hashes / no-whitespace
3792 // CJK runs all hit this. The fix hard-breaks overlong words at
3793 // grapheme boundaries; these tests pin that across widths 40/60/80/120.
3794
3795 fn rendered_widths(rendered: &[Line<'static>]) -> Vec<usize> {
3796 rendered
3797 .iter()
3798 .map(|l| {
3799 l.spans
3800 .iter()
3801 .map(|s| s.content.as_ref().width())
3802 .sum::<usize>()
3803 })
3804 .collect()
3805 }
3806
3807 fn render_paragraph_for_test(text: &str, width: usize) -> Vec<Line<'static>> {
3808 render_line_with_links(text, width, Style::default(), Style::default())
3809 }
3810
3811 #[test]
3812 fn paragraph_wrap_breaks_overlong_word_at_width_40() {
3813 // 200-char no-whitespace token must not exceed the 40-col window.
3814 let long = "a".repeat(200);
3815 let rendered = render_paragraph_for_test(&long, 40);
3816 for w in rendered_widths(&rendered) {
3817 assert!(w <= 40, "rendered width {w} exceeds 40-col window");
3818 }
3819 // And the full content must still be present across the wrapped lines.
3820 let combined: String = rendered
3821 .iter()
3822 .flat_map(|l| l.spans.iter().map(|s| s.content.to_string()))
3823 .collect();
3824 assert_eq!(combined.matches('a').count(), 200);
3825 }
3826
3827 #[test]
3828 fn paragraph_wrap_breaks_no_whitespace_cjk_at_width_40() {
3829 // #963: long CJK runs without whitespace must wrap by display width
3830 // instead of overflowing or truncating. Each Han character is 2 cols.
3831 let long = "界".repeat(300);
3832 let rendered = render_paragraph_for_test(&long, 40);
3833 for w in rendered_widths(&rendered) {
3834 assert!(w <= 40, "rendered width {w} exceeds 40-col window");
3835 }
3836 let combined: String = rendered
3837 .iter()
3838 .flat_map(|l| l.spans.iter().map(|s| s.content.to_string()))
3839 .collect();
3840 assert_eq!(combined.chars().filter(|&ch| ch == '界').count(), 300);
3841 assert!(
3842 rendered.len() >= 15,
3843 "300 double-width chars should wrap into many rows, got {}",
3844 rendered.len()
3845 );
3846 }
3847
3848 #[test]
3849 fn paragraph_wrap_breaks_overlong_word_at_widths_60_80_120() {
3850 let long = format!("https://example.com/{}", "p".repeat(180));
3851 for &width in &[60usize, 80, 120] {
3852 let rendered = render_paragraph_for_test(&long, width);
3853 for w in rendered_widths(&rendered) {
3854 assert!(
3855 w <= width,
3856 "width={width}: rendered line width {w} exceeds budget"
3857 );
3858 }
3859 assert!(rendered.len() >= 2, "width={width}: expected wrap");
3860 }
3861 }
3862
3863 #[test]
3864 fn paragraph_wrap_keeps_short_words_unbroken() {
3865 // Regression guard: short words must still be broken at whitespace,
3866 // not mid-word. Width 40, only short words, expect zero mid-word
3867 // breaks (each line reads as natural English).
3868 let text = "the quick brown fox jumps over the lazy dog and then it stops moving";
3869 let rendered = render_paragraph_for_test(text, 40);
3870 for line in &rendered {
3871 let s: String = line.spans.iter().map(|s| s.content.to_string()).collect();
3872 // Heuristic: trimmed line should not start with a partial word
3873 // (i.e. should start with a real English start) — every line in
3874 // this fixture starts with a word in our short list.
3875 let first = s.split_whitespace().next().unwrap_or("");
3876 assert!(
3877 [
3878 "the", "quick", "brown", "fox", "jumps", "over", "lazy", "dog", "and", "then",
3879 "it", "stops", "moving"
3880 ]
3881 .contains(&first),
3882 "line {s:?} appears to start with a partial word"
3883 );
3884 }
3885 }
3886
3887 #[test]
3888 fn paragraph_wrap_mixed_short_and_overlong_word() {
3889 // The overlong word must wrap; the trailing short words must pack
3890 // onto subsequent lines. The combined content is preserved.
3891 let long = "x".repeat(150);
3892 let text = format!("intro {long} tail words go here");
3893 let rendered = render_paragraph_for_test(&text, 80);
3894 for w in rendered_widths(&rendered) {
3895 assert!(w <= 80, "rendered width {w} exceeds 80-col window");
3896 }
3897 let combined: String = rendered
3898 .iter()
3899 .flat_map(|l| l.spans.iter().map(|s| s.content.to_string()))
3900 .collect();
3901 for fragment in ["intro", "tail", "words", "go", "here"] {
3902 assert!(
3903 combined.contains(fragment),
3904 "fragment {fragment:?} missing from wrapped output:\n{combined}"
3905 );
3906 }
3907 assert_eq!(combined.matches('x').count(), 150);
3908 }
3909
3910 #[test]
3911 fn wrap_text_breaks_overlong_word_for_code_blocks() {
3912 // The standalone code-block wrap (wrap_text) had the same overflow
3913 // bug; pin the fix at widths 40 and 80.
3914 for &width in &[40usize, 80] {
3915 let long = "z".repeat(200);
3916 let lines = wrap_text(&long, width);
3917 for line in &lines {
3918 assert!(
3919 line.width() <= width,
3920 "wrap_text line {line:?} exceeds {width}"
3921 );
3922 }
3923 let combined: String = lines.join("");
3924 assert_eq!(combined.matches('z').count(), 200);
3925 }
3926 }
3927
3928 #[test]
3929 fn wrap_cell_text_already_handled_long_words_remains_correct() {
3930 // Regression guard for the v0.8.25 table-cell fix. After consolidating
3931 // the char-break helper, wrap_cell_text must continue to handle
3932 // overlong cells. Pin the property: every wrapped segment fits
3933 // within the column width, and content is preserved.
3934 let long = "y".repeat(120);
3935 let segments = wrap_cell_text(&long, 30);
3936 for seg in &segments {
3937 assert!(seg.width() <= 30, "segment {seg:?} exceeds col 30");
3938 }
3939 let combined: String = segments.join("");
3940 assert_eq!(combined.matches('y').count(), 120);
3941 }
3942
3943 #[test]
3944 fn paragraph_wrap_handles_zero_width_gracefully() {
3945 // Width 0 should not panic or hang; it returns the input as-is or
3946 // empty, but never produces a line wider than 0 (when 0 means "no
3947 // budget at all"). This pins the early-return path against future
3948 // regressions.
3949 let rendered = render_paragraph_for_test("hello world", 0);
3950 // Any output is acceptable (the path is degenerate); assert no panic.
3951 let _ = rendered;
3952 }
3953
3954 fn rendered_text(rendered: &[Line<'static>]) -> String {
3955 rendered
3956 .iter()
3957 .flat_map(|l| l.spans.iter().map(|s| s.content.as_ref()))
3958 .collect()
3959 }
3960
3961 fn assert_rendered_widths_fit(rendered: &[Line<'static>], width: usize, label: &str) {
3962 for line_width in rendered_widths(rendered) {
3963 assert!(
3964 line_width <= width,
3965 "{label} width={width}: rendered line width {line_width} exceeds budget"
3966 );
3967 }
3968 }
3969
3970 // ── Unicode / CJK / emoji / combining-char width QA (#3488) ────────────
3971
3972 #[test]
3973 fn paragraph_wrap_keeps_unicode_runs_within_qa_widths() {
3974 let cases = [
3975 ("cjk", "界".repeat(300)),
3976 ("emoji", "😀".repeat(200)),
3977 ("mixed-cjk-emoji", "界😀世🚀".repeat(90)),
3978 ];
3979
3980 for (label, text) in cases {
3981 for &width in &[80usize, 100, 120] {
3982 let rendered = render_paragraph_for_test(&text, width);
3983 assert_rendered_widths_fit(&rendered, width, label);
3984 assert_eq!(
3985 rendered_text(&rendered),
3986 text,
3987 "{label} width={width}: content changed while wrapping"
3988 );
3989
3990 let min_lines = text.width().div_ceil(width);
3991 assert!(
3992 rendered.len() >= min_lines,
3993 "{label} width={width}: expected at least {min_lines} lines, got {}",
3994 rendered.len()
3995 );
3996 }
3997 }
3998 }
3999
4000 #[test]
4001 fn paragraph_wrap_preserves_mixed_unicode_and_ascii_fragments() {
4002 let cjk = "这是一个测试字符串".repeat(10); // 80 Han chars = 160 cols
4003 let emoji = "🚀".repeat(12);
4004 let text = format!("Note: {cjk} done {emoji}");
4005
4006 for &width in &[80usize, 100, 120] {
4007 let rendered = render_paragraph_for_test(&text, width);
4008 assert_rendered_widths_fit(&rendered, width, "mixed unicode/ascii");
4009
4010 let visible = visible_lines(&rendered).join("\n");
4011 for fragment in ["Note:", "测试", "done"] {
4012 assert!(
4013 visible.contains(fragment),
4014 "width={width}: fragment {fragment:?} missing from output:\n{visible}"
4015 );
4016 }
4017 assert_eq!(
4018 visible.matches('🚀').count(),
4019 12,
4020 "width={width}: emoji content lost"
4021 );
4022 }
4023 }
4024
4025 #[test]
4026 fn lower_level_wrap_text_keeps_unicode_runs_within_qa_widths() {
4027 let cases = [
4028 ("cjk", "中".repeat(140)),
4029 ("emoji", "😀".repeat(110)),
4030 ("combining", "e\u{301}".repeat(140)),
4031 ];
4032
4033 for (label, input) in cases {
4034 for &width in &[80usize, 100, 120] {
4035 let lines = wrap_text(&input, width);
4036 for line in &lines {
4037 assert!(
4038 line.width() <= width,
4039 "{label} width={width}: wrap_text line {line:?} exceeds budget"
4040 );
4041 }
4042 let combined: String = lines.join("");
4043 assert_eq!(
4044 combined, input,
4045 "{label} width={width}: wrap_text changed content"
4046 );
4047 }
4048 }
4049 }
4050
4051 #[test]
4052 fn table_render_keeps_cjk_cells_within_qa_widths() {
4053 let cjk = "界".repeat(80);
4054 let src = format!("| Name | Value |\n|---|---|\n| CJK | {cjk} |\n");
4055
4056 for &width in &[80usize, 100, 120] {
4057 let rendered = render_markdown(&src, width as u16, Style::default());
4058 assert_rendered_widths_fit(&rendered, width, "table cjk");
4059 assert_eq!(
4060 rendered_text(&rendered).matches('界').count(),
4061 80,
4062 "width={width}: CJK table cell content lost"
4063 );
4064 }
4065 }
4066
4067 #[test]
4068 fn paragraph_wrap_keeps_cjk_transcript_within_narrow_widths() {
4069 // The seed cases cover 80/100/120; narrow terminals (resize / small
4070 // panes) are the other half of #3488's terminal-width lane. A CJK
4071 // transcript paragraph must still wrap inside tiny windows without
4072 // overflowing the border or dropping content.
4073 let text = "实时输出结果显示正常".repeat(6); // 60 Han glyphs, 120 cols
4074 for &width in &[20usize, 40] {
4075 let rendered = render_paragraph_for_test(&text, width);
4076 assert_rendered_widths_fit(&rendered, width, "narrow cjk transcript");
4077 assert_eq!(
4078 rendered_text(&rendered),
4079 text,
4080 "width={width}: CJK transcript content changed while wrapping"
4081 );
4082 }
4083 }
4084
4085 // A deliberate measurement probe, not an assertion: run it with
4086 // `--nocapture` to read the per-append cost. Printing is the whole point,
4087 // so the module-wide stdout ban is lifted here the same way
4088 // `core/engine/tests.rs` lifts it for its probes.
4089 #[allow(clippy::print_stdout)]
4090 #[test]
4091 fn probe_incremental_stream_cost() {
4092 // Faithful to one streaming message: the document grows a word at a
4093 // time and the render path re-parses and re-renders the whole visible
4094 // text on every tick. Reports the cost per 250 appends so growth with
4095 // message size is visible.
4096 use std::time::Instant;
4097 let words: Vec<&str> =
4098 "the quick brown fox jumps over a lazy dog and then keeps going for a while"
4099 .split(' ')
4100 .collect();
4101 let mut doc = String::new();
4102 let width = 100u16;
4103 let mut window_start = Instant::now();
4104 for i in 0..2000usize {
4105 doc.push_str(words[i % words.len()]);
4106 doc.push(' ');
4107 let parsed = parse(&doc);
4108 let lines = render_parsed(&parsed, width, Style::default());
4109 std::hint::black_box(&lines);
4110 if (i + 1) % 250 == 0 {
4111 let elapsed = window_start.elapsed();
4112 println!(
4113 "PROBE bytes={} per_append={:?} total={:?}",
4114 doc.len(),
4115 elapsed / 250,
4116 elapsed
4117 );
4118 window_start = Instant::now();
4119 }
4120 }
4121 }
4122 }
4123
4123 lines RUST