返回 CodeWhale
latex_render.rs
根目录 / crates / tui / src / tui / history / latex_render.rs
1 //! LaTeX math expression rendering for the TUI transcript.
2 //! Renders `$...$` (inline) and `$$...$$` (display) math expressions using
3 //! Unicode approximations for terminal display.
4 use std::collections::HashMap;
5 use std::sync::OnceLock;
6 use unicode_width::UnicodeWidthStr;
7
8 fn is_escaped(bytes: &[u8], idx: usize) -> bool {
9 let mut slashes = 0;
10 let mut cursor = idx;
11 while cursor > 0 && bytes[cursor - 1] == b'\\' {
12 slashes += 1;
13 cursor -= 1;
14 }
15 slashes % 2 == 1
16 }
17 /// Private: find the start of math delimiter ($, $$, \(, \[) in text.
18 fn find_math_start(text: &str) -> Option<usize> {
19 let b = text.as_bytes();
20 for (idx, &byte) in b.iter().enumerate() {
21 if byte == b'$' && !is_escaped(b, idx) {
22 return Some(idx);
23 }
24 if byte == b'\\'
25 && !is_escaped(b, idx)
26 && idx + 1 < b.len()
27 && (b[idx + 1] == b'(' || b[idx + 1] == b'[')
28 {
29 return Some(idx);
30 }
31 }
32 None
33 }
34 /// Private: if text starts with a math delimiter, find closing delimiter and return (end_pos, is_display).
35 fn find_math_end(text: &str) -> Option<(usize, bool)> {
36 let b = text.as_bytes();
37 if b.starts_with(b"$$") {
38 for i in 2..b.len().saturating_sub(1) {
39 if b[i] == b'{' {
40 continue;
41 }
42 if b[i..].starts_with(b"$$") && !is_escaped(b, i) {
43 return Some((i, true));
44 }
45 }
46 } else if b.starts_with(b"$") && !b.starts_with(b"$$") {
47 for (j, &byte) in b[1..].iter().enumerate() {
48 if byte == b'{' {
49 continue;
50 }
51 let i = j + 1;
52 if byte == b'$'
53 && !is_escaped(b, i)
54 && !b
55 .get(i.wrapping_sub(1))
56 .is_some_and(u8::is_ascii_whitespace)
57 && !b.get(i + 1).is_some_and(u8::is_ascii_digit)
58 {
59 return Some((i, false));
60 }
61 }
62 } else if b.starts_with(b"\\[") {
63 for i in 2..b.len().saturating_sub(1) {
64 if b[i] == b'{' {
65 continue;
66 }
67 if b[i..].starts_with(b"\\]") && !is_escaped(b, i) {
68 return Some((i, true));
69 }
70 }
71 } else if b.starts_with(b"\\(") {
72 for i in 2..b.len().saturating_sub(1) {
73 if b[i] == b'{' {
74 continue;
75 }
76 if b[i..].starts_with(b"\\)") && !is_escaped(b, i) {
77 return Some((i, false));
78 }
79 }
80 }
81 None
82 }
83 fn math_delim_offset(text: &str) -> usize {
84 let b = text.as_bytes();
85 if b.starts_with(b"$$") || b.starts_with(b"\\[") || b.starts_with(b"\\(") {
86 2
87 } else {
88 1
89 }
90 }
91 fn render_math_segment(text: &str) -> String {
92 let mut result = String::new();
93 let mut i = 0;
94 while i < text.len() {
95 let remaining = &text[i..];
96 if let Some((end, _is_display)) = find_math_end(remaining) {
97 let offset = math_delim_offset(remaining);
98 let inner = &remaining[offset..end];
99 result.push_str(&render_latex_to_string(inner));
100 let close_len: usize = if remaining[end..].starts_with("\\]")
101 || remaining[end..].starts_with("$$")
102 || remaining[end..].starts_with("\\)")
103 {
104 2
105 } else if remaining.as_bytes().get(end..end + 1) == Some(b"$") {
106 1
107 } else {
108 0
109 };
110 i += end + close_len;
111 } else {
112 let skip = find_math_start(remaining).unwrap_or(remaining.len());
113 if skip == 0 {
114 // Unmatched opening delimiter ($, $$, \(, \[) during streaming:
115 // push it as plain text so the loop can advance.
116 result.push(remaining.chars().next().unwrap_or('$'));
117 i += remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(1);
118 } else {
119 result.push_str(&remaining[..skip]);
120 i += skip;
121 if skip == remaining.len() {
122 break;
123 }
124 }
125 }
126 }
127 result
128 }
129
130 /// Replace math delimiters with plain Unicode while preserving Markdown code.
131 ///
132 /// Fast path (#perf-r5): the overwhelming majority of streamed content
133 /// contains no math delimiters at all. A single byte scan for the three
134 /// opening delimiters (`$`, `\(`, `\[`) decides between borrowing the input
135 /// untouched and running the full transform, so the per-chunk streaming
136 /// render avoids allocating a full-content copy on every update when no
137 /// math is present.
138 pub fn render_latex_in_text(text: &str) -> std::borrow::Cow<'_, str> {
139 // Math can only start at '$' (incl. '$$') or the two-byte '\(' and '\['.
140 // Scanning bytes directly avoids a regex; any hit falls back to the
141 // full transform below, which re-verifies delimiters precisely.
142 let has_delim = text
143 .as_bytes()
144 .iter()
145 .enumerate()
146 .any(|(idx, &byte)| match byte {
147 b'$' => true,
148 b'\\' => matches!(text.as_bytes().get(idx + 1), Some(b'(') | Some(b'[')),
149 _ => false,
150 });
151 if !has_delim {
152 return std::borrow::Cow::Borrowed(text);
153 }
154 let mut result = String::with_capacity(text.len());
155 let mut cursor = 0;
156
157 while cursor < text.len() {
158 let Some(tick_offset) = text[cursor..].find('`') else {
159 result.push_str(&render_math_segment(&text[cursor..]));
160 break;
161 };
162 let tick_start = cursor + tick_offset;
163 result.push_str(&render_math_segment(&text[cursor..tick_start]));
164
165 let tick_count = text[tick_start..]
166 .bytes()
167 .take_while(|byte| *byte == b'`')
168 .count();
169 let delimiter = "`".repeat(tick_count);
170 let content_start = tick_start + tick_count;
171 if let Some(close_offset) = text[content_start..].find(&delimiter) {
172 let code_end = content_start + close_offset + tick_count;
173 result.push_str(&text[tick_start..code_end]);
174 cursor = code_end;
175 } else {
176 result.push_str(&text[tick_start..]);
177 break;
178 }
179 }
180
181 std::borrow::Cow::Owned(result)
182 }
183
184 // --- Environment rendering ---
185
186 /// Render a `\begin{name}...\end{name}` block.
187 /// `content` is everything between the braces.
188 fn render_environment(env_name: &str, content: &str) -> String {
189 match env_name {
190 "aligned" | "align" | "gather" | "eqnarray" | "split" => render_aligned(content),
191 "pmatrix" | "bmatrix" | "vmatrix" | "Bmatrix" | "matrix" | "smallmatrix" => {
192 render_matrix(env_name, content)
193 }
194 "array" => render_array(content),
195 "cases" | "dcases" => render_cases(content, false),
196 "rcases" | "drcases" => render_cases(content, true),
197 _ => {
198 // Unknown environment: pass through raw
199 format!("\\begin{{{env_name}}}{content}\\end{{{env_name}}}")
200 }
201 }
202 }
203
204 /// Split a multi-row env content into rows (`\\` separator), each rendered.
205 /// `row_fn` is called for each parsed row (list of cell strings).
206 fn parse_rows<F>(content: &str, mut row_fn: F)
207 where
208 F: FnMut(Vec<String>),
209 {
210 // Split by \\ (but be careful: \\\\ is an escaped backslash, not a line break)
211 let mut current = String::new();
212 let mut chars = content.chars().peekable();
213 while let Some(ch) = chars.next() {
214 if ch == '\\' {
215 if chars.peek() == Some(&'\\') {
216 // Line break marker
217 chars.next(); // consume second \
218 // If followed by optional whitespace and an optional * (\\*)
219 while matches!(chars.peek(), Some(&' ') | Some(&'\t')) {
220 chars.next();
221 }
222 if chars.peek() == Some(&'*') {
223 chars.next();
224 }
225 row_fn(parse_row_cells(&current));
226 current.clear();
227 } else if chars.peek() == Some(&'[') || chars.peek() == Some(&'{') {
228 // --- Spacing ---
229 // --- Spacing ---
230 if chars.peek() == Some(&'[') {
231 chars.next();
232 while let Some(&c) = chars.peek() {
233 if c == ']' {
234 chars.next();
235 break;
236 }
237 chars.next();
238 }
239 } else if chars.peek() == Some(&'{') {
240 let _ = read_braced_chars(&mut chars);
241 }
242 row_fn(parse_row_cells(&current));
243 current.clear();
244 } else {
245 current.push('\\');
246 }
247 } else if ch == '\n' {
248 // Newlines in environments often act as row separators
249 // but not inside braces
250 // Simple approach: treat bare \n as space
251 if !current.is_empty() && !current.ends_with(' ') {
252 current.push(' ');
253 }
254 } else {
255 current.push(ch);
256 }
257 }
258 // Last row
259 row_fn(parse_row_cells(&current));
260 }
261
262 /// Split a single row into cells by `&`.
263 fn parse_row_cells(row: &str) -> Vec<String> {
264 // Split by & but skip escaped \&
265 let mut cells = Vec::new();
266 let mut current = String::new();
267 let mut chars = row.chars().peekable();
268 while let Some(ch) = chars.next() {
269 if ch == '&' {
270 cells.push(current.trim().to_string());
271 current.clear();
272 } else if ch == '\\' && chars.peek() == Some(&'&') {
273 // Escaped ampersand
274 current.push('&');
275 chars.next();
276 } else {
277 current.push(ch);
278 }
279 }
280 cells.push(current.trim().to_string());
281 cells
282 }
283
284 /// Aligned equations: align at `&` markers.
285 fn render_aligned(content: &str) -> String {
286 let mut rows: Vec<Vec<String>> = Vec::new();
287 parse_rows(content, |cells| rows.push(cells));
288
289 if rows.is_empty() {
290 return String::new();
291 }
292
293 // Determine max columns
294 let max_cols = rows.iter().map(|r| r.len()).max().unwrap_or(0);
295 if max_cols == 0 {
296 return String::new();
297 }
298
299 // Double-pass: render each cell and measure widths
300 let mut rendered: Vec<Vec<String>> = Vec::new();
301 let mut col_widths: Vec<usize> = vec![0; max_cols];
302
303 for row in &rows {
304 let mut rendered_row = Vec::new();
305 for (ci, cell) in row.iter().enumerate() {
306 let rendered_cell = render_latex_to_string(cell);
307 let w = UnicodeWidthStr::width(rendered_cell.as_str());
308 if ci < max_cols && w > col_widths[ci] {
309 col_widths[ci] = w;
310 }
311 rendered_row.push(rendered_cell);
312 }
313 // Pad missing cells
314 while rendered_row.len() < max_cols {
315 rendered_row.push(String::new());
316 }
317 rendered.push(rendered_row);
318 }
319
320 // Second pass: assemble with padding
321 let mut result = String::new();
322 for (ri, row) in rendered.iter().enumerate() {
323 if ri > 0 {
324 result.push('\n');
325 }
326 for ci in 0..max_cols {
327 if ci > 0 {
328 let pad =
329 col_widths[ci - 1].saturating_sub(UnicodeWidthStr::width(row[ci - 1].as_str()));
330 for _ in 0..pad {
331 result.push(' ');
332 }
333 result.push_str(" ");
334 }
335 result.push_str(&row[ci]);
336 }
337 }
338
339 result
340 }
341
342 /// --- Brackets ---
343 fn render_matrix(env_name: &str, content: &str) -> String {
344 let mut rows: Vec<Vec<String>> = Vec::new();
345 parse_rows(content, |cells| rows.push(cells));
346
347 if rows.is_empty() {
348 return String::new();
349 }
350
351 let max_cols = rows.iter().map(|r| r.len()).max().unwrap_or(0);
352 if max_cols == 0 {
353 return String::new();
354 }
355
356 // Two-pass: render + measure
357 let mut rendered: Vec<Vec<String>> = Vec::new();
358 let mut col_widths: Vec<usize> = vec![0; max_cols];
359
360 for row in &rows {
361 let mut rendered_row = Vec::new();
362 for (ci, cell) in row.iter().enumerate() {
363 let rendered_cell = render_latex_to_string(cell);
364 let w = UnicodeWidthStr::width(rendered_cell.as_str());
365 if ci < max_cols && w > col_widths[ci] {
366 col_widths[ci] = w;
367 }
368 rendered_row.push(rendered_cell);
369 }
370 while rendered_row.len() < max_cols {
371 rendered_row.push(String::new());
372 }
373 rendered.push(rendered_row);
374 }
375
376 // Build each row with proper padding
377 let mut cell_strings: Vec<String> = Vec::new();
378 for row in &rendered {
379 let mut line = String::new();
380 for ci in 0..max_cols {
381 if ci > 0 {
382 line.push(' ');
383 }
384 let cell = &row[ci];
385 line.push_str(cell);
386 let pad = col_widths[ci].saturating_sub(UnicodeWidthStr::width(cell.as_str()));
387 for _ in 0..pad {
388 line.push(' ');
389 }
390 }
391 cell_strings.push(line);
392 }
393
394 match env_name {
395 "pmatrix" => surround_with("(", ")", &cell_strings, 1),
396 "bmatrix" => surround_with("[", "]", &cell_strings, 1),
397 "vmatrix" => surround_with("\u{2502}", "\u{2502}", &cell_strings, 1),
398 "Bmatrix" => surround_with("{", "}", &cell_strings, 1),
399 "smallmatrix" => surround_with("(", ")", &cell_strings, 0),
400 _ => {
401 // --- Brackets ---
402 let mut result = String::new();
403 for (ri, s) in cell_strings.iter().enumerate() {
404 if ri > 0 {
405 result.push('\n');
406 }
407 result.push_str(s);
408 }
409 result
410 }
411 }
412 }
413
414 /// --- Brackets ---
415 fn surround_with(left: &str, right: &str, rows: &[String], pad: usize) -> String {
416 if rows.is_empty() {
417 return format!("{left}{right}");
418 }
419 let mut result = String::new();
420 if rows.len() == 1 {
421 result.push_str(left);
422 for _ in 0..pad {
423 result.push(' ');
424 }
425 result.push_str(&rows[0]);
426 for _ in 0..pad {
427 result.push(' ');
428 }
429 result.push_str(right);
430 return result;
431 }
432 // Multi-row: brackets on their own lines
433 result.push_str(left);
434 result.push('\n');
435 for (ri, s) in rows.iter().enumerate() {
436 if ri > 0 {
437 result.push('\n');
438 }
439 for _ in 0..pad {
440 result.push(' ');
441 }
442 result.push_str(s);
443 }
444 result.push('\n');
445 result.push_str(right);
446 result
447 }
448
449 /// Array environment: parse column spec and render table with vertical bars.
450 fn render_array(content: &str) -> String {
451 let trimmed = content.trim_start();
452 let (col_spec, body) = if let Some(after_brace) = trimmed.strip_prefix('{') {
453 let close = after_brace.find('}').map(|i| i + 1).unwrap_or(0);
454 if close > 0 {
455 (&after_brace[..close], after_brace[close..].trim_start())
456 } else {
457 ("", trimmed)
458 }
459 } else {
460 ("", trimmed)
461 };
462 let mut has_vline_start = false;
463 let mut vlines = Vec::new();
464 let mut cols = Vec::new();
465 for ch in col_spec.chars() {
466 match ch {
467 'c' | 'l' | 'r' => cols.push(ch),
468 '|' => {
469 if cols.is_empty() {
470 has_vline_start = true;
471 } else {
472 vlines.push(cols.len());
473 }
474 }
475 _ => {}
476 }
477 }
478 let n = cols.len();
479 if n == 0 {
480 return body.to_string();
481 }
482 let mut rows = Vec::new();
483 parse_rows(body, |cells| rows.push(cells));
484 let mut result = String::new();
485 for (ri, row) in rows.iter().enumerate() {
486 if ri > 0 {
487 result.push('\n');
488 }
489 if has_vline_start {
490 result.push_str("| ");
491 }
492 for ci in 0..n {
493 if ci > 0 {
494 result.push(if vlines.contains(&ci) { '|' } else { ' ' });
495 result.push(' ');
496 }
497 result.push_str(&render_latex_to_string(
498 row.get(ci).unwrap_or(&String::new()),
499 ));
500 }
501 if vlines.contains(&n) || has_vline_start {
502 result.push_str(" |");
503 }
504 }
505 result
506 }
507
508 /// Piecewise functions with cases environment.
509 fn render_cases(content: &str, right_brace: bool) -> String {
510 let mut rows: Vec<Vec<String>> = Vec::new();
511 parse_rows(content, |cells| rows.push(cells));
512
513 if rows.is_empty() {
514 return String::new();
515 }
516
517 let mut rendered_rows: Vec<(String, Option<String>)> = Vec::new();
518 let mut left_width = 0;
519
520 for cells in &rows {
521 let left = render_latex_to_string(cells.first().map(|s| s.as_str()).unwrap_or(""));
522 let left_w = UnicodeWidthStr::width(left.as_str());
523 if left_w > left_width {
524 left_width = left_w;
525 }
526 let right = if cells.len() > 1 {
527 let r = render_latex_to_string(&cells[1]);
528 Some(r)
529 } else {
530 None
531 };
532 rendered_rows.push((left, right));
533 }
534
535 let n = rendered_rows.len();
536 let mut result = String::new();
537 for (ri, (left, right)) in rendered_rows.iter().enumerate() {
538 if ri > 0 {
539 result.push('\n');
540 }
541 if !right_brace {
542 result.push_str(match ri {
543 0 => "\u{23a7} ",
544 _ if ri == n - 1 => "\u{23a9} ",
545 _ => "\u{23a8} ",
546 });
547 }
548
549 // Left part + padding
550 let left_pad = left_width.saturating_sub(UnicodeWidthStr::width(left.as_str()));
551 result.push_str(left);
552 for _ in 0..left_pad {
553 result.push(' ');
554 }
555
556 if let Some(cond) = right {
557 result.push_str(", ");
558 result.push_str(cond);
559 }
560 }
561 result
562 }
563
564 // --- Helper: read_braced for chars iterator ---
565
566 fn read_braced_chars(chars: &mut std::iter::Peekable<std::str::Chars>) -> String {
567 let mut s = String::new();
568 let mut depth: u32 = 0;
569 if chars.next_if_eq(&'{').is_some() {
570 depth = 1;
571 }
572 while let Some(&c) = chars.peek() {
573 match c {
574 '{' => {
575 depth += 1;
576 s.push(c);
577 chars.next();
578 }
579 '}' => {
580 depth = depth.saturating_sub(1);
581 chars.next();
582 if depth == 0 {
583 break;
584 }
585 s.push('}');
586 }
587 _ => {
588 s.push(c);
589 chars.next();
590 }
591 }
592 }
593 s
594 }
595
596 // --- Styled symbols ---
597
598 fn render_styled_symbol(
599 command: &str,
600 chars: &mut std::iter::Peekable<std::str::Chars>,
601 out: &mut String,
602 ) {
603 let argument = read_braced_chars(chars);
604 let rendered = match (command, argument.as_str()) {
605 ("mathbb", "R") => Some("\u{211d}"),
606 ("mathbb", "C") => Some("\u{2102}"),
607 ("mathbb", "N") => Some("\u{2115}"),
608 ("mathbb", "Q") => Some("\u{211a}"),
609 ("mathbb", "Z") => Some("\u{2124}"),
610 ("mathbb", "P") => Some("\u{2119}"),
611 ("mathbb", "H") => Some("\u{210d}"),
612 ("mathbb", "F") => Some("\u{1d53b}"),
613 ("mathcal", "L") => Some("\u{2112}"),
614 ("mathcal", "H") => Some("\u{210b}"),
615 ("mathcal", "R") => Some("\u{211b}"),
616 ("mathcal", "A") => Some("\u{1d49c}"),
617 ("mathcal", "B") => Some("\u{212c}"),
618 ("mathcal", "C") => Some("\u{212d}"),
619 ("mathcal", "D") => Some("\u{1d49f}"),
620 ("mathcal", "E") => Some("\u{2130}"),
621 ("mathcal", "F") => Some("\u{2131}"),
622 ("mathcal", "I") => Some("\u{2110}"),
623 ("mathcal", "M") => Some("\u{2133}"),
624 ("mathcal", "O") => Some("\u{1d4aa}"),
625 ("mathcal", "P") => Some("\u{1d4ab}"),
626 ("mathcal", "S") => Some("\u{1d4ae}"),
627 ("mathcal", "T") => Some("\u{1d4af}"),
628 ("mathcal", "Z") => Some("\u{2128}"),
629 _ => None,
630 };
631 if let Some(symbol) = rendered {
632 out.push_str(symbol);
633 } else {
634 out.push('\\');
635 out.push_str(command);
636 out.push('{');
637 out.push_str(&argument);
638 out.push('}');
639 }
640 }
641
642 // --- Main render function ---
643
644 /// Read a braced group `{...}` from an index-based cursor in a string.
645 /// Returns (content_string, new_cursor_position).
646 fn read_braced_at(input: &str, start: usize) -> Option<(String, usize)> {
647 let bytes = input.as_bytes();
648 let mut pos = start;
649 if pos >= input.len() || bytes[pos] != b'{' {
650 return None;
651 }
652 pos += 1; // skip {
653 let mut depth: u32 = 1;
654 let mut content = String::new();
655 while pos < input.len() {
656 let ch = input[pos..].chars().next()?;
657 let byte_len = ch.len_utf8();
658 match ch {
659 '{' => {
660 depth += 1;
661 if depth > 1 {
662 content.push('{');
663 }
664 }
665 '}' => {
666 depth -= 1;
667 if depth == 0 {
668 return Some((content, pos + 1));
669 }
670 content.push('}');
671 }
672 _ => content.push(ch),
673 }
674 pos += byte_len;
675 }
676 None
677 }
678
679 /// Main LaTeX-to-Unicode rendering.
680 fn render_latex_to_string(latex: &str) -> String {
681 let input = latex.trim();
682 let mut out = String::new();
683 let mut pos = 0;
684
685 while pos < input.len() {
686 let remaining = &input[pos..];
687
688 // 1. Environment detection: \begin{name}...\end{name}
689 if let Some(env_rendered) = try_render_env(remaining) {
690 let (rendered, consumed) = env_rendered;
691 // Start multi-line envs on a fresh line for alignment
692 let needs_newline = rendered.starts_with('\u{23a7}') // cases ??
693 || rendered.starts_with('(') // pmatrix
694 || rendered.starts_with('[') // bmatrix
695 || rendered.starts_with('\u{2502}'); // vmatrix
696 if needs_newline && !out.ends_with('\n') {
697 out.push('\n');
698 }
699 out.push_str(&rendered);
700 pos += consumed;
701 continue;
702 }
703
704 let ch = remaining.chars().next().unwrap();
705 let ch_len = ch.len_utf8();
706
707 match ch {
708 '\\' => {
709 // Read command name
710 let rest = &input[pos + 1..];
711 let cmd_end = rest
712 .find(|c: char| !c.is_ascii_alphabetic())
713 .unwrap_or(rest.len());
714 let cmd = &rest[..cmd_end];
715 let _after_cmd = cmd_end;
716
717 if cmd.is_empty() {
718 // Escape sequence for special chars
719 if let Some(&next) = rest.as_bytes().first() {
720 match next {
721 b'{' | b'}' | b'$' | b'%' | b'#' | b'&' | b'_' | b' ' => {
722 // Consume the escape
723 let skip = 1 + 1; // \ + char
724 if next == b' ' {
725 // \ (backslash-space) is a space
726 out.push(' ');
727 }
728 // otherwise just skip (it's an escaped char)
729 pos += skip;
730 continue;
731 }
732 _ => {
733 // Unknown single-char escape 闁?output the char
734 let char_len =
735 rest.chars().next().map(|c| c.len_utf8()).unwrap_or(1);
736 out.push(rest.chars().next().unwrap_or(ch));
737 pos += 1 + char_len;
738 continue;
739 }
740 }
741 }
742 pos += 1;
743 continue;
744 }
745
746 let cmd_len = cmd.len();
747 let _total_cmd_start = pos;
748 let total_cmd_end = pos + 1 + cmd_len; // \ + name
749
750 match cmd {
751 // --- Environments (handled above, just in case) ---
752 "begin" => {
753 // Should have been caught by try_render_env above.
754 // Fallback: try inline parsing
755 if let Some((env_name, after_name)) = read_braced_at(input, total_cmd_end) {
756 let end_tag = format!("\\end{{{env_name}}}");
757 let search_from = &input[after_name..];
758 if let Some(end_rel) = search_from.find(&end_tag) {
759 let env_content = &search_from[..end_rel];
760 let rendered = render_environment(&env_name, env_content);
761 out.push_str(&rendered);
762 pos = after_name + end_rel + end_tag.len();
763 continue;
764 }
765 out.push_str(&format!("\\begin{{{env_name}}}"));
766 }
767 out.push_str("\\begin");
768 pos = total_cmd_end;
769 }
770 "end" => {
771 // Shouldn't be reached; output raw.
772 let after_end = pos + 4;
773 if let Some((env_name, after_name)) = read_braced_at(input, after_end) {
774 out.push_str(&format!("\\end{{{env_name}}}"));
775 pos = after_name;
776 } else {
777 out.push_str("\\end");
778 pos = after_end;
779 }
780 }
781 // --- Text ---
782 "text" | "mathrm" | "mathit" | "mathsf" | "textrm" | "textit" | "textbf" => {
783 if let Some((arg, new_pos)) = read_braced_at(input, total_cmd_end) {
784 out.push_str(&arg);
785 pos = new_pos;
786 } else {
787 // No braces, try single char
788 let next = &input[total_cmd_end..].chars().next();
789 if let Some(c) = next {
790 out.push(*c);
791 pos = total_cmd_end + c.len_utf8();
792 } else {
793 out.push_str(cmd);
794 pos = total_cmd_end;
795 }
796 }
797 }
798 // --- Accents ---
799 "hat" | "bar" | "tilde" | "dot" | "ddot" | "vec" | "breve" | "check"
800 | "acute" | "grave" => {
801 if let Some((arg, new_pos)) = read_braced_at(input, total_cmd_end) {
802 let rendered = render_latex_to_string(&arg);
803 let accent = match cmd {
804 "hat" => "\u{0302}",
805 "bar" => "\u{0304}",
806 "tilde" => "\u{0303}",
807 "dot" => "\u{0307}",
808 "ddot" => "\u{0308}",
809 "vec" => "\u{20d7}",
810 "breve" => "\u{0306}",
811 "check" => "\u{030c}",
812 "acute" => "\u{0301}",
813 "grave" => "\u{0300}",
814 _ => unreachable!(),
815 };
816 out.push_str(&rendered);
817 out.push_str(accent);
818 pos = new_pos;
819 } else {
820 out.push_str(&format!("\\{cmd}"));
821 pos = total_cmd_end;
822 }
823 }
824 "operatorname" => {
825 if let Some((arg, new_pos)) = read_braced_at(input, total_cmd_end) {
826 out.push_str(&arg);
827 pos = new_pos;
828 } else {
829 out.push_str("\\operatorname");
830 pos = total_cmd_end;
831 }
832 }
833 // --- Underbrace / Overbrace (passthrough) ---
834 "underbrace" | "overbrace" | "underbracket" | "overbracket" => {
835 if let Some((arg, after_arg)) = read_braced_at(input, total_cmd_end) {
836 out.push_str(&render_latex_to_string(&arg));
837 pos = after_arg;
838 } else {
839 out.push_str(&format!("\\{cmd}"));
840 pos = total_cmd_end;
841 }
842 }
843 // --- Vertical/horizontal phantom (no-op) ---
844 "vphantom" | "hphantom" | "phantom" => {
845 if let Some((_, after_arg)) = read_braced_at(input, total_cmd_end) {
846 pos = after_arg;
847 } else {
848 out.push_str(&format!("\\{cmd}"));
849 pos = total_cmd_end;
850 }
851 }
852 // --- Substack ---
853 "substack" => {
854 if let Some((arg, after_arg)) = read_braced_at(input, total_cmd_end) {
855 out.push_str(&arg);
856 pos = after_arg;
857 } else {
858 out.push_str(&format!("\\{cmd}"));
859 pos = total_cmd_end;
860 }
861 }
862 // --- Fonts ---
863 "mathbf" | "bf" => {
864 if let Some((arg, new_pos)) = read_braced_at(input, total_cmd_end) {
865 out.push_str(&render_latex_to_string(&arg));
866 pos = new_pos;
867 } else {
868 let next = &input[total_cmd_end..].chars().next();
869 if let Some(c) = next {
870 out.push(*c);
871 pos = total_cmd_end + c.len_utf8();
872 } else {
873 out.push_str(cmd);
874 pos = total_cmd_end;
875 }
876 }
877 }
878 // --- Brackets ---
879 "left" | "bigl" | "Bigl" | "biggl" | "Biggl" => {
880 // Consume the next token (bracket/pipe/dot) and output it
881 let after = &input[total_cmd_end..].trim_start();
882 if let Some(next) = after.chars().next() {
883 if next == '.' {
884 // \left. 闁?invisible delimiter, skip
885 } else {
886 out.push(next);
887 }
888 let skip = after.len() - after.trim_start().len() + next.len_utf8();
889 pos = total_cmd_end + skip;
890 } else {
891 pos = total_cmd_end;
892 }
893 }
894 "right" | "bigr" | "Bigr" | "biggr" | "Biggr" => {
895 let after = &input[total_cmd_end..].trim_start();
896 if let Some(next) = after.chars().next() {
897 if next == '.' {
898 // \right. 闁?invisible delimiter, skip
899 } else {
900 out.push(next);
901 }
902 let skip = after.len() - after.trim_start().len() + next.len_utf8();
903 pos = total_cmd_end + skip;
904 } else {
905 pos = total_cmd_end;
906 }
907 }
908 "big" | "Big" | "bigg" | "Bigg" => {
909 // Size modifiers 闁?skip them, the next token is what matters
910 let after = &input[total_cmd_end..].trim_start();
911 if let Some(next) = after.chars().next() {
912 out.push(next);
913 pos = total_cmd_end
914 + (after.len() - after.trim_start().len())
915 + next.len_utf8();
916 } else {
917 pos = total_cmd_end;
918 }
919 }
920 // --- Spacing ---
921 "quad" => {
922 out.push_str(" ");
923 pos = total_cmd_end;
924 }
925 "qquad" => {
926 out.push_str(" ");
927 pos = total_cmd_end;
928 }
929 "," | "thinspace" => {
930 out.push(' ');
931 pos = total_cmd_end;
932 }
933 ";" | "thickspace" => {
934 out.push_str(" ");
935 pos = total_cmd_end;
936 }
937 "!" | "negthinspace" => {
938 // Negative space: just skip
939 pos = total_cmd_end;
940 }
941 ":" | "medspace" => {
942 out.push_str(" ");
943 pos = total_cmd_end;
944 }
945 " " | "space" | "enspace" => {
946 out.push(' ');
947 pos = total_cmd_end;
948 }
949 // --- Styled symbols ---
950 "mathbb" | "mathcal" => {
951 let before = out.len();
952 if let Some((arg, after_arg)) = read_braced_at(input, total_cmd_end) {
953 let mut chars = arg.chars().peekable();
954 render_styled_symbol(cmd, &mut chars, &mut out);
955 if out.len() == before {
956 // render_styled_symbol didn't match
957 out.push_str(&format!("\\{cmd}{{{arg}}}"));
958 }
959 pos = after_arg;
960 } else {
961 out.push_str(&format!("\\{cmd}"));
962 pos = total_cmd_end;
963 }
964 }
965 // --- Fractions ---
966 "frac" | "dfrac" | "tfrac" | "cfrac" => {
967 if let Some((num_s, after_num)) = read_braced_at(input, total_cmd_end) {
968 if let Some((den_s, after_den)) = read_braced_at(input, after_num) {
969 let n = render_latex_to_string(&num_s);
970 let d = render_latex_to_string(&den_s);
971 out.push_str(&format!("({n}/{d})"));
972 pos = after_den;
973 } else {
974 out.push_str(&format!("({num_s}/?)"));
975 pos = after_num;
976 }
977 } else {
978 out.push_str(&format!("\\{cmd}"));
979 pos = total_cmd_end;
980 }
981 }
982 // --- Binomial coefficient ---
983 "binom" => {
984 if let Some((top_s, after_top)) = read_braced_at(input, total_cmd_end)
985 && let Some((bot_s, after_bot)) = read_braced_at(input, after_top)
986 {
987 out.push_str(&format!(
988 "({}/{})",
989 render_latex_to_string(&top_s),
990 render_latex_to_string(&bot_s)
991 ));
992 pos = after_bot;
993 } else {
994 // Malformed: keep the command literal and advance,
995 // or the loop would re-read it forever.
996 out.push_str("\\binom");
997 pos = total_cmd_end;
998 }
999 }
1000 // --- Square root ---
1001 "sqrt" => {
1002 let after = &input[total_cmd_end..];
1003 // Optional [n] root index, without its brackets.
1004 let (root_text, after_root) = match after
1005 .strip_prefix('[')
1006 .and_then(|after_lb| after_lb.find(']').map(|end| (after_lb, end)))
1007 {
1008 Some((after_lb, end)) => {
1009 (Some(&after_lb[..end]), total_cmd_end + end + 2)
1010 }
1011 None => (None, total_cmd_end),
1012 };
1013 if let Some(root) = root_text {
1014 append_superscript(&render_latex_to_string(root), &mut out);
1015 }
1016 out.push('\u{221a}');
1017 // One radical, and resume after the whole argument:
1018 // the old path pushed a second radical and re-read the
1019 // argument, and a radicand without braces never
1020 // advanced at all (U04-m2).
1021 if let Some((arg, after_arg)) = read_braced_at(input, after_root) {
1022 out.push_str(&format!("({})", render_latex_to_string(&arg)));
1023 pos = after_arg;
1024 } else {
1025 pos = after_root;
1026 }
1027 }
1028 // --- Sum, product, integral ---
1029 "sum" => {
1030 out.push('\u{2211}');
1031 pos = total_cmd_end;
1032 }
1033 "prod" => {
1034 out.push('\u{220f}');
1035 pos = total_cmd_end;
1036 }
1037 "int" => {
1038 out.push('\u{222b}');
1039 pos = total_cmd_end;
1040 }
1041 "iint" => {
1042 out.push('\u{222c}');
1043 pos = total_cmd_end;
1044 }
1045 "iiint" => {
1046 out.push('\u{222d}');
1047 pos = total_cmd_end;
1048 }
1049 "oint" => {
1050 out.push('\u{222e}');
1051 pos = total_cmd_end;
1052 }
1053 "oiint" => {
1054 out.push('\u{222f}');
1055 pos = total_cmd_end;
1056 }
1057 // --- Named operators ---
1058 "lim" => {
1059 out.push_str("lim");
1060 pos = total_cmd_end;
1061 }
1062 "sin" | "cos" | "tan" | "cot" | "sec" | "csc" | "log" | "ln" | "lg" | "exp"
1063 | "det" | "dim" | "ker" | "hom" | "max" | "min" | "sup" | "inf" | "arg"
1064 | "deg" | "mod" | "gcd" | "lcm" | "Pr" | "Var" | "Cov" | "Corr" | "tr"
1065 | "rank" | "Re" | "Im" | "sinh" | "cosh" | "tanh" | "coth" | "arcsin"
1066 | "arccos" | "arctan" => {
1067 out.push_str(cmd);
1068 pos = total_cmd_end;
1069 }
1070 // --- Arrows ---
1071 "to" | "rightarrow" => {
1072 out.push('\u{2192}');
1073 pos = total_cmd_end;
1074 }
1075 "leftarrow" => {
1076 out.push('\u{2190}');
1077 pos = total_cmd_end;
1078 }
1079 "Rightarrow" => {
1080 out.push('\u{21d2}');
1081 pos = total_cmd_end;
1082 }
1083 "Leftarrow" => {
1084 out.push('\u{21d0}');
1085 pos = total_cmd_end;
1086 }
1087 "Leftrightarrow" | "iff" => {
1088 out.push('\u{21d4}');
1089 pos = total_cmd_end;
1090 }
1091 "mapsto" => {
1092 out.push('\u{21a6}');
1093 pos = total_cmd_end;
1094 }
1095 "longrightarrow" => {
1096 out.push('\u{27f6}');
1097 pos = total_cmd_end;
1098 }
1099 "Longrightarrow" => {
1100 out.push('\u{27f9}');
1101 pos = total_cmd_end;
1102 }
1103 "uparrow" => {
1104 out.push('\u{2191}');
1105 pos = total_cmd_end;
1106 }
1107 "downarrow" => {
1108 out.push('\u{2193}');
1109 pos = total_cmd_end;
1110 }
1111 "Uparrow" => {
1112 out.push('\u{21d1}');
1113 pos = total_cmd_end;
1114 }
1115 "Downarrow" => {
1116 out.push('\u{21d3}');
1117 pos = total_cmd_end;
1118 }
1119 "longleftrightarrow" => {
1120 out.push('\u{27f7}');
1121 pos = total_cmd_end;
1122 }
1123 "Longleftrightarrow" => {
1124 out.push('\u{27fa}');
1125 pos = total_cmd_end;
1126 }
1127 "hookrightarrow" => {
1128 out.push('\u{21aa}');
1129 pos = total_cmd_end;
1130 }
1131 "hookleftarrow" => {
1132 out.push('\u{21a9}');
1133 pos = total_cmd_end;
1134 }
1135 "rightharpoonup" => {
1136 out.push('\u{21c0}');
1137 pos = total_cmd_end;
1138 }
1139 "rightharpoondown" => {
1140 out.push('\u{21c1}');
1141 pos = total_cmd_end;
1142 }
1143 "leftharpoonup" => {
1144 out.push('\u{21bc}');
1145 pos = total_cmd_end;
1146 }
1147 "leftharpoondown" => {
1148 out.push('\u{21bd}');
1149 pos = total_cmd_end;
1150 }
1151 "rightleftharpoons" => {
1152 out.push('\u{21cc}');
1153 pos = total_cmd_end;
1154 }
1155 "nrightarrow" => {
1156 out.push('\u{219b}');
1157 pos = total_cmd_end;
1158 }
1159 "nleftarrow" => {
1160 out.push('\u{219a}');
1161 pos = total_cmd_end;
1162 }
1163 // --- Unknown command ---
1164 _ => {
1165 if let Some(sym) = SYMBOLS.get_or_init(build_symbols).get(cmd) {
1166 out.push_str(sym);
1167 pos = total_cmd_end;
1168 // Check for braces after symbol (e.g., \alpha_{i})
1169 // The subscript/superscript will be handled by the
1170 // main loop as _ and ^
1171 } else {
1172 // --- Unknown command ---
1173 out.push('\\');
1174 out.push_str(cmd);
1175 pos = total_cmd_end;
1176 // If followed by {, include the braced argument
1177 if input[pos..].starts_with('{')
1178 && let Some((arg, new_pos)) = read_braced_at(input, pos)
1179 {
1180 out.push('{');
1181 out.push_str(&arg);
1182 out.push('}');
1183 pos = new_pos;
1184 }
1185 }
1186 }
1187 }
1188 }
1189 '_' => {
1190 // Read subscript
1191 let after = &input[pos + 1..];
1192 if after.starts_with('{') {
1193 if let Some((sub, new_pos)) = read_braced_at(input, pos + 1) {
1194 append_subscript(&render_latex_to_string(&sub), &mut out);
1195 pos = new_pos;
1196 } else {
1197 out.push('_');
1198 pos += 1;
1199 }
1200 } else {
1201 // Subscript with command like _\mu _\nu
1202 if let Some(after_bs) = after.strip_prefix('\\') {
1203 let cmd_end = after_bs
1204 .find(|c: char| !c.is_ascii_alphabetic())
1205 .unwrap_or(after_bs.len());
1206 let rendered = render_latex_to_string(&after[..1 + cmd_end]);
1207 append_subscript(&rendered, &mut out);
1208 pos += 1 + 1 + cmd_end;
1209 } else {
1210 let next = after.chars().next();
1211 if let Some(c) = next {
1212 append_subscript(&c.to_string(), &mut out);
1213 pos += 1 + c.len_utf8();
1214 } else {
1215 out.push('_');
1216 pos += 1;
1217 }
1218 }
1219 }
1220 }
1221 '^' => {
1222 // Read superscript
1223 let after = &input[pos + 1..];
1224 if after.starts_with('{') {
1225 if let Some((sup, new_pos)) = read_braced_at(input, pos + 1) {
1226 append_superscript(&render_latex_to_string(&sup), &mut out);
1227 pos = new_pos;
1228 } else {
1229 out.push('^');
1230 pos += 1;
1231 }
1232 } else {
1233 // Superscript with command like ^\dagger ^\rho
1234 if let Some(after_bs) = after.strip_prefix('\\') {
1235 let cmd_end = after_bs
1236 .find(|c: char| !c.is_ascii_alphabetic())
1237 .unwrap_or(after_bs.len());
1238 let rendered = render_latex_to_string(&after[..1 + cmd_end]);
1239 append_superscript(&rendered, &mut out);
1240 pos += 1 + 1 + cmd_end;
1241 } else {
1242 let next = after.chars().next();
1243 if let Some(c) = next {
1244 append_superscript(&c.to_string(), &mut out);
1245 pos += 1 + c.len_utf8();
1246 } else {
1247 out.push('^');
1248 pos += 1;
1249 }
1250 }
1251 }
1252 }
1253 '{' | '}' => {
1254 pos += ch_len;
1255 }
1256 ' ' => {
1257 if !out.ends_with(' ') {
1258 out.push(' ');
1259 }
1260 pos += ch_len;
1261 }
1262 '\n' => {
1263 if !out.ends_with(' ') {
1264 out.push(' ');
1265 }
1266 pos += ch_len;
1267 }
1268 // Punctuation that shouldn't be duplicated
1269 '~' => {
1270 // Non-breaking space
1271 out.push(' ');
1272 pos += ch_len;
1273 }
1274 _ => {
1275 out.push(ch);
1276 pos += ch_len;
1277 }
1278 }
1279 }
1280
1281 out.trim_end().to_string()
1282 }
1283
1284 /// Try to parse a `\begin{env_name}...\end{env_name}` block at the start of `input`.
1285 /// Returns (rendered_output, bytes_consumed) or None.
1286 fn try_render_env(input: &str) -> Option<(String, usize)> {
1287 let input_bytes = input.as_bytes();
1288
1289 // Check for \begin{
1290 if input.len() < 7 || &input_bytes[..7] != b"\\begin{" {
1291 return None;
1292 }
1293
1294 // Find closing }
1295 let close = input[7..].find('}')?;
1296 let env_name = &input[7..7 + close];
1297
1298 let content_start = 7 + close + 1; // after \begin{env_name}
1299 if content_start >= input.len() {
1300 return None;
1301 }
1302
1303 // Find matching \end{env_name}
1304 let end_tag = format!("\\end{{{env_name}}}");
1305 let rest = &input[content_start..];
1306
1307 // Simple depth tracking for nested braces
1308 let mut depth = 0i32;
1309 let mut search_pos = 0;
1310
1311 while search_pos < rest.len() {
1312 let remaining_search = &rest[search_pos..];
1313
1314 if remaining_search.starts_with(&end_tag) && depth == 0 {
1315 let env_content = &rest[..search_pos];
1316 let rendered = render_environment(env_name, env_content);
1317 let consumed = content_start + search_pos + end_tag.len();
1318 // --- Spacing ---
1319 return Some((rendered, consumed));
1320 }
1321
1322 match remaining_search.as_bytes().first()? {
1323 b'{' => depth += 1,
1324 b'}' => depth -= 1,
1325 _ => {}
1326 }
1327 search_pos += 1;
1328 }
1329
1330 None
1331 }
1332
1333 // --- Superscript / Subscript ---
1334
1335 fn append_superscript(s: &str, out: &mut String) {
1336 for c in s.chars() {
1337 out.push(match c {
1338 '0' => '\u{2070}',
1339 '1' => '\u{00b9}',
1340 '2' => '\u{00b2}',
1341 '3' => '\u{00b3}',
1342 '4' => '\u{2074}',
1343 '5' => '\u{2075}',
1344 '6' => '\u{2076}',
1345 '7' => '\u{2077}',
1346 '8' => '\u{2078}',
1347 '9' => '\u{2079}',
1348 '+' => '\u{207a}',
1349 '-' => '\u{207b}',
1350 '=' => '\u{207c}',
1351 '(' => '\u{207d}',
1352 ')' => '\u{207e}',
1353 'n' => '\u{207f}',
1354 'i' => '\u{2071}',
1355 'a' => '\u{1d43}',
1356 'b' => '\u{1d47}',
1357 'c' => '\u{1d9c}',
1358 'd' => '\u{1d48}',
1359 'e' => '\u{1d49}',
1360 'f' => '\u{1da0}',
1361 'g' => '\u{1d4d}',
1362 'h' => '\u{02b0}',
1363 'j' => '\u{02b2}',
1364 'k' => '\u{1d4f}',
1365 'l' => '\u{02e1}',
1366 'm' => '\u{1d50}',
1367 'o' => '\u{1d52}',
1368 'p' => '\u{1d56}',
1369 'r' => '\u{02b3}',
1370 's' => '\u{02e2}',
1371 't' => '\u{1d57}',
1372 'u' => '\u{1d58}',
1373 'v' => '\u{1d5b}',
1374 'w' => '\u{02b7}',
1375 'x' => '\u{02e3}',
1376 'y' => '\u{02b8}',
1377 'z' => '\u{1dbb}',
1378 _ => c,
1379 });
1380 }
1381 }
1382
1383 fn append_subscript(s: &str, out: &mut String) {
1384 for c in s.chars() {
1385 out.push(match c {
1386 '0' => '\u{2080}',
1387 '1' => '\u{2081}',
1388 '2' => '\u{2082}',
1389 '3' => '\u{2083}',
1390 '4' => '\u{2084}',
1391 '5' => '\u{2085}',
1392 '6' => '\u{2086}',
1393 '7' => '\u{2087}',
1394 '8' => '\u{2088}',
1395 '9' => '\u{2089}',
1396 '+' => '\u{208a}',
1397 '-' => '\u{208b}',
1398 '=' => '\u{208c}',
1399 'a' => '\u{2090}',
1400 'e' => '\u{2091}',
1401 'h' => '\u{2095}',
1402 'i' => '\u{1d62}',
1403 'k' => '\u{2096}',
1404 'l' => '\u{2097}',
1405 'm' => '\u{2098}',
1406 'n' => '\u{2099}',
1407 'o' => '\u{2092}',
1408 'p' => '\u{209a}',
1409 'r' => '\u{1d63}',
1410 's' => '\u{209b}',
1411 't' => '\u{209c}',
1412 'u' => '\u{1d64}',
1413 'v' => '\u{1d65}',
1414 'x' => '\u{2093}',
1415 _ => c,
1416 });
1417 }
1418 }
1419
1420 // --- Symbol table ---
1421
1422 type SymbolMap = HashMap<&'static str, &'static str>;
1423 fn build_symbols() -> SymbolMap {
1424 let mut m = SymbolMap::new();
1425 // Lowercase Greek
1426 for (k, v) in [
1427 ("alpha", "\u{03b1}"),
1428 ("beta", "\u{03b2}"),
1429 ("gamma", "\u{03b3}"),
1430 ("delta", "\u{03b4}"),
1431 ("epsilon", "\u{03b5}"),
1432 ("zeta", "\u{03b6}"),
1433 ("eta", "\u{03b7}"),
1434 ("theta", "\u{03b8}"),
1435 ("iota", "\u{03b9}"),
1436 ("kappa", "\u{03ba}"),
1437 ("lambda", "\u{03bb}"),
1438 ("mu", "\u{03bc}"),
1439 ("nu", "\u{03bd}"),
1440 ("xi", "\u{03be}"),
1441 ("pi", "\u{03c0}"),
1442 ("rho", "\u{03c1}"),
1443 ("sigma", "\u{03c3}"),
1444 ("tau", "\u{03c4}"),
1445 ("upsilon", "\u{03c5}"),
1446 ("phi", "\u{03c6}"),
1447 ("chi", "\u{03c7}"),
1448 ("psi", "\u{03c8}"),
1449 ("omega", "\u{03c9}"),
1450 ("varepsilon", "\u{03b5}"),
1451 ("vartheta", "\u{03d1}"),
1452 ("varphi", "\u{03c6}"),
1453 ("varrho", "\u{03f1}"),
1454 ] {
1455 m.insert(k, v);
1456 }
1457 // Uppercase Greek
1458 for (k, v) in [
1459 ("Gamma", "\u{0393}"),
1460 ("Delta", "\u{0394}"),
1461 ("Theta", "\u{0398}"),
1462 ("Lambda", "\u{039b}"),
1463 ("Xi", "\u{039e}"),
1464 ("Pi", "\u{03a0}"),
1465 ("Sigma", "\u{03a3}"),
1466 ("Upsilon", "\u{03a5}"),
1467 ("Phi", "\u{03a6}"),
1468 ("Psi", "\u{03a8}"),
1469 ("Omega", "\u{03a9}"),
1470 ] {
1471 m.insert(k, v);
1472 }
1473 // Miscellaneous
1474 for (k, v) in [
1475 ("infty", "\u{221e}"),
1476 ("partial", "\u{2202}"),
1477 ("nabla", "\u{2207}"),
1478 ("ell", "\u{2113}"),
1479 ("hbar", "\u{210f}"),
1480 ("Im", "\u{2111}"),
1481 ("Re", "\u{211c}"),
1482 ("emptyset", "\u{2205}"),
1483 ("varnothing", "\u{2205}"),
1484 ("aleph", "\u{2135}"),
1485 ("angle", "\u{2220}"),
1486 ("measuredangle", "\u{2221}"),
1487 ("langle", "\u{27e8}"),
1488 ("rangle", "\u{27e9}"),
1489 ("perp", "\u{22a5}"),
1490 ("parallel", "\u{2225}"),
1491 ("nparallel", "\u{2226}"),
1492 ("prime", "\u{2032}"),
1493 ("surd", "\u{221a}"),
1494 ("top", "\u{22a4}"),
1495 ("bot", "\u{22a5}"),
1496 ("imath", "\u{0131}"),
1497 ("jmath", "\u{0237}"),
1498 ("wp", "\u{2118}"),
1499 ("clubsuit", "\u{2663}"),
1500 ("diamondsuit", "\u{2662}"),
1501 ("heartsuit", "\u{2661}"),
1502 ("spadesuit", "\u{2660}"),
1503 ("triangle", "\u{25b3}"),
1504 ("Box", "\u{25a1}"),
1505 ("Diamond", "\u{25c7}"),
1506 ("flat", "\u{266d}"),
1507 ("natural", "\u{266e}"),
1508 ("sharp", "\u{266f}"),
1509 ("colon", ":"),
1510 ("backslash", "\\"),
1511 ] {
1512 m.insert(k, v);
1513 }
1514 // Set / relation symbols
1515 for (k, v) in [
1516 ("in", "\u{2208}"),
1517 ("notin", "\u{2209}"),
1518 ("ni", "\u{220b}"),
1519 ("subset", "\u{2282}"),
1520 ("supset", "\u{2283}"),
1521 ("subseteq", "\u{2286}"),
1522 ("supseteq", "\u{2287}"),
1523 ("subsetneq", "\u{228a}"),
1524 ("supsetneq", "\u{228b}"),
1525 ("cup", "\u{222a}"),
1526 ("bigcup", "\u{22c3}"),
1527 ("cap", "\u{2229}"),
1528 ("bigcap", "\u{22c2}"),
1529 ("vee", "\u{2228}"),
1530 ("wedge", "\u{2227}"),
1531 ("oplus", "\u{2295}"),
1532 ("ominus", "\u{2296}"),
1533 ("otimes", "\u{2297}"),
1534 ("oslash", "\u{2298}"),
1535 ("odot", "\u{2299}"),
1536 ("sqcap", "\u{2293}"),
1537 ("sqcup", "\u{2294}"),
1538 ("uplus", "\u{228e}"),
1539 ("amalg", "\u{2a3f}"),
1540 ("forall", "\u{2200}"),
1541 ("exists", "\u{2203}"),
1542 ("nexists", "\u{2204}"),
1543 ("neg", "\u{00ac}"),
1544 ("lnot", "\u{00ac}"),
1545 ("land", "\u{2227}"),
1546 ("lor", "\u{2228}"),
1547 ("implies", "\u{21d2}"),
1548 ("iff", "\u{21d4}"),
1549 ("gets", "\u{2190}"),
1550 ("sim", "\u{223c}"),
1551 ("nsim", "\u{2241}"),
1552 ("simeq", "\u{2243}"),
1553 ("nsimeq", "\u{2244}"),
1554 ("cong", "\u{2245}"),
1555 ("ncong", "\u{2247}"),
1556 ("approx", "\u{2248}"),
1557 ("napprox", "\u{2249}"),
1558 ("neq", "\u{2260}"),
1559 ("ne", "\u{2260}"),
1560 ("equiv", "\u{2261}"),
1561 ("nequiv", "\u{2262}"),
1562 ("le", "\u{2264}"),
1563 ("ge", "\u{2265}"),
1564 ("leq", "\u{2264}"),
1565 ("geq", "\u{2265}"),
1566 ("leqq", "\u{2266}"),
1567 ("geqq", "\u{2267}"),
1568 ("lneq", "\u{2268}"),
1569 ("gneq", "\u{2269}"),
1570 ("ll", "\u{226a}"),
1571 ("gg", "\u{226b}"),
1572 ("lll", "\u{22d8}"),
1573 ("ggg", "\u{22d9}"),
1574 ("prec", "\u{227a}"),
1575 ("succ", "\u{227b}"),
1576 ("preceq", "\u{227c}"),
1577 ("succeq", "\u{227d}"),
1578 ("preccurlyeq", "\u{227c}"),
1579 ("succcurlyeq", "\u{227d}"),
1580 ("propto", "\u{221d}"),
1581 ("models", "\u{22a7}"),
1582 ("dashv", "\u{22a3}"),
1583 ("vdash", "\u{22a2}"),
1584 ("mid", "|"),
1585 ("nmid", "\u{2224}"),
1586 ] {
1587 m.insert(k, v);
1588 }
1589 // Operators
1590 for (k, v) in [
1591 ("times", "\u{00d7}"),
1592 ("div", "\u{00f7}"),
1593 ("pm", "\u{00b1}"),
1594 ("mp", "\u{2213}"),
1595 ("cdot", "\u{00b7}"),
1596 ("ast", "\u{2217}"),
1597 ("circ", "\u{2218}"),
1598 ("bullet", "\u{2022}"),
1599 ("setminus", "\u{2216}"),
1600 ("smallsetminus", "\u{2216}"),
1601 ("wr", "\u{2240}"),
1602 ("dagger", "\u{2020}"),
1603 ("ddagger", "\u{2021}"),
1604 ("star", "\u{22c6}"),
1605 ("diamond", "\u{22c4}"),
1606 ] {
1607 m.insert(k, v);
1608 }
1609 // Dots
1610 for (k, v) in [
1611 ("cdots", "\u{2026}"),
1612 ("ldots", "\u{2026}"),
1613 ("vdots", "\u{22ee}"),
1614 ("ddots", "\u{22f1}"),
1615 ("idots", "\u{2026}"),
1616 ] {
1617 m.insert(k, v);
1618 }
1619 // Named functions not covered by the inline list
1620 for (k, v) in [
1621 ("arccos", "arccos"),
1622 ("arcsin", "arcsin"),
1623 ("arctan", "arctan"),
1624 ("arg", "arg"),
1625 ("cos", "cos"),
1626 ("cosh", "cosh"),
1627 ("cot", "cot"),
1628 ("coth", "coth"),
1629 ("csc", "csc"),
1630 ("deg", "deg"),
1631 ("det", "det"),
1632 ("dim", "dim"),
1633 ("exp", "exp"),
1634 ("gcd", "gcd"),
1635 ("hom", "hom"),
1636 ("inf", "inf"),
1637 ("ker", "ker"),
1638 ("lg", "lg"),
1639 ("lim", "lim"),
1640 ("liminf", "liminf"),
1641 ("limsup", "limsup"),
1642 ("ln", "ln"),
1643 ("log", "log"),
1644 ("max", "max"),
1645 ("min", "min"),
1646 ("mod", "mod"),
1647 ("sec", "sec"),
1648 ("sin", "sin"),
1649 ("sinh", "sinh"),
1650 ("sup", "sup"),
1651 ("tan", "tan"),
1652 ("tanh", "tanh"),
1653 ] {
1654 m.insert(k, v);
1655 }
1656 m
1657 }
1658
1659 static SYMBOLS: OnceLock<SymbolMap> = OnceLock::new();
1660
1661 #[cfg(test)]
1662 mod tests {
1663 use super::*;
1664
1665 #[test]
1666 fn test_superscript() {
1667 assert_eq!(render_latex_to_string("x^2"), "x\u{00b2}");
1668 }
1669 /// U04-m2: one radical, the index as a superscript, and rendering
1670 /// resumes after the argument instead of printing it twice.
1671 #[test]
1672 fn sqrt_renders_one_radical_and_consumes_its_argument() {
1673 assert_eq!(render_latex_to_string(r"\sqrt{x}+1"), "\u{221a}(x)+1");
1674 assert_eq!(
1675 render_latex_to_string(r"\sqrt[3]{x}"),
1676 "\u{00b3}\u{221a}(x)"
1677 );
1678 }
1679
1680 /// A radicand or binomial without braces must still finish rendering;
1681 /// the old branches never advanced and spun the render loop forever.
1682 #[test]
1683 fn malformed_sqrt_and_binom_terminate() {
1684 let (tx, rx) = std::sync::mpsc::channel();
1685 std::thread::spawn(move || {
1686 let _ = tx.send((
1687 render_latex_to_string(r"\sqrt 2"),
1688 render_latex_to_string(r"\binom{n} k"),
1689 ));
1690 });
1691 let (sqrt, binom) = rx
1692 .recv_timeout(std::time::Duration::from_secs(5))
1693 .expect("latex rendering must terminate");
1694 assert!(sqrt.starts_with('\u{221a}'), "{sqrt}");
1695 assert!(sqrt.ends_with('2'), "{sqrt}");
1696 assert!(binom.contains("binom"), "{binom}");
1697 }
1698
1699 #[test]
1700 fn test_subscript() {
1701 assert_eq!(render_latex_to_string("x_1"), "x\u{2081}");
1702 }
1703 #[test]
1704 fn test_blackboard() {
1705 assert_eq!(render_latex_to_string(r"\mathbb{R}"), "\u{211d}");
1706 }
1707 #[test]
1708 fn test_infty() {
1709 assert_eq!(render_latex_to_string(r"\infty"), "\u{221e}");
1710 }
1711 #[test]
1712 fn test_inline_dollar() {
1713 let r = render_latex_in_text(r"text $x^2$ more");
1714 assert_eq!(r, "text x\u{00b2} more");
1715 assert!(matches!(r, std::borrow::Cow::Owned(_)));
1716 }
1717 #[test]
1718 fn test_display_bracket() {
1719 let r = render_latex_in_text(r"text \[x^2\] more");
1720 assert_eq!(r, "text x\u{00b2} more");
1721 }
1722 #[test]
1723 fn no_math_is_borrowed_without_copy() {
1724 let r = render_latex_in_text("plain prose with `code` but no math at all");
1725 assert!(matches!(r, std::borrow::Cow::Borrowed(_)));
1726 assert_eq!(&*r, "plain prose with `code` but no math at all");
1727 // The '$' fast path must not miss \(
1728 let p = render_latex_in_text("parens \\(x^2\\) inline");
1729 assert!(matches!(p, std::borrow::Cow::Owned(_)));
1730 assert_eq!(&*p, "parens x\u{00b2} inline");
1731 }
1732 #[test]
1733 fn preserves_currency() {
1734 assert_eq!(render_latex_in_text("cost $5 and $10"), "cost $5 and $10");
1735 }
1736 #[test]
1737 fn preserves_markdown_code() {
1738 assert_eq!(
1739 render_latex_in_text("`$x^2$` and $y^2$"),
1740 "`$x^2$` and y\u{00b2}"
1741 );
1742 assert_eq!(
1743 render_latex_in_text("```sh\necho $HOME\n```"),
1744 "```sh\necho $HOME\n```"
1745 );
1746 }
1747 #[test]
1748 fn preserves_escaped_dollars_and_unknown_commands() {
1749 assert_eq!(
1750 render_latex_in_text(r"cost \$5 and $\operatorname{foo}$"),
1751 r"cost \$5 and foo"
1752 );
1753 }
1754 #[test]
1755 fn test_text() {
1756 assert_eq!(render_latex_to_string(r"\text{hello}"), "hello");
1757 }
1758 #[test]
1759 fn test_operatorname() {
1760 assert_eq!(render_latex_to_string(r"\operatorname{sgn}"), "sgn");
1761 }
1762 #[test]
1763 fn test_left_right() {
1764 assert_eq!(
1765 render_latex_to_string(r"\left(\frac{a}{b}\right)"),
1766 "((a/b))"
1767 );
1768 }
1769 #[test]
1770 fn test_mathbf() {
1771 assert_eq!(render_latex_to_string(r"\mathbf{E}"), "E");
1772 }
1773 #[test]
1774 fn test_quad() {
1775 assert_eq!(render_latex_to_string(r"a \quad b"), "a b");
1776 }
1777 #[test]
1778 fn test_environment_aligned() {
1779 let input = r"\begin{aligned} x &= y \\ a &= b \end{aligned}";
1780 let result = render_latex_to_string(input);
1781 assert!(result.contains("x"));
1782 assert!(result.contains("y"));
1783 assert!(result.contains("a"));
1784 assert!(result.contains("b"));
1785 }
1786 #[test]
1787 fn test_environment_matrix() {
1788 let input = r"\begin{pmatrix} a & b \\ c & d \end{pmatrix}";
1789 let result = render_latex_to_string(input);
1790 assert!(result.contains("a"));
1791 assert!(result.contains("b"));
1792 assert!(result.contains("c"));
1793 assert!(result.contains("d"));
1794 }
1795 #[test]
1796 fn test_environment_cases() {
1797 let input = r"\begin{cases} x^2 & x < 0 \\ 0 & x = 0 \\ \ln x & x > 0 \end{cases}";
1798 let result = render_latex_to_string(input);
1799 assert!(result.contains("x\u{00b2}"));
1800 assert!(result.contains("ln"));
1801 }
1802 }
1803
1803 lines RUST