| 1 | // Golden-case verification for the math rendering pipeline. |
| 2 | // |
| 3 | // Run: tsx src/__tests__/math-golden.test.ts |
| 4 | // |
| 5 | // We import the *production* modules (mathNormalize, latexNormalize, |
| 6 | // mathClassify) rather than reimplementing them inline, so this file |
| 7 | // catches regressions in the actual code path that runs inside <Markdown>. |
| 8 | |
| 9 | import { createElement } from "react"; |
| 10 | import { renderToStaticMarkup } from "react-dom/server"; |
| 11 | import ReactMarkdown from "react-markdown"; |
| 12 | import katex from "katex"; |
| 13 | import { latexNormalizeForKatex, stripMathDelimiters } from "../components/latexNormalize"; |
| 14 | import { classifyInlineMath, isLikelyInlineMath } from "../components/mathClassify"; |
| 15 | import { reasonixRehypePlugins, reasonixRemarkPlugins } from "../components/markdownRemarkPlugins"; |
| 16 | import { |
| 17 | normalizeMath, |
| 18 | restoreProtectedInlineMathSource, |
| 19 | } from "../components/mathNormalize"; |
| 20 | |
| 21 | let passed = 0; |
| 22 | let failed = 0; |
| 23 | |
| 24 | function check(label: string, fn: () => boolean) { |
| 25 | try { |
| 26 | if (fn()) { process.stdout.write(` PASS ${label}\n`); passed += 1; } |
| 27 | else { process.stdout.write(` FAIL ${label}\n`); failed += 1; } |
| 28 | } catch (e) { |
| 29 | process.stdout.write(` ERROR ${label}: ${(e as Error).message}\n`); failed += 1; |
| 30 | } |
| 31 | } |
| 32 | |
| 33 | function eq(a: unknown, b: unknown, label: string) { |
| 34 | if (a === b) { |
| 35 | process.stdout.write(` PASS ${label}\n`); |
| 36 | passed += 1; |
| 37 | } else { |
| 38 | process.stdout.write(` FAIL ${label}: expected ${JSON.stringify(b)}, got ${JSON.stringify(a)}\n`); |
| 39 | failed += 1; |
| 40 | } |
| 41 | } |
| 42 | |
| 43 | // ── stripMathDelimiters ──────────────────────────────────────────────────────── |
| 44 | |
| 45 | console.log("\nstripMathDelimiters"); |
| 46 | eq(stripMathDelimiters("\\(x+1\\)"), "x+1", "\\(...\\)"); |
| 47 | eq(stripMathDelimiters("\\[E=mc^2\\]"), "E=mc^2", "\\[...\\]"); |
| 48 | eq(stripMathDelimiters("$$\\frac{a}{b}$$"), "\\frac{a}{b}", "$$...$$"); |
| 49 | eq(stripMathDelimiters("$x_i^2$"), "x_i^2", "$...$"); |
| 50 | eq(stripMathDelimiters("plain text"), "plain text", "no delimiters"); |
| 51 | eq(stripMathDelimiters("$a|b$"), "a|b", "inline with pipe"); |
| 52 | |
| 53 | // ── latexNormalizeForKatex ───────────────────────────────────────────────────── |
| 54 | |
| 55 | console.log("\nlatexNormalizeForKatex"); |
| 56 | eq(latexNormalizeForKatex("x+1"), "x+1", "plain unchanged"); |
| 57 | eq(latexNormalizeForKatex("\\text{baryon #}"), "\\text{baryon \\#}", "escapes # in \\text"); |
| 58 | eq(latexNormalizeForKatex("\\text{cost is $5}"), "\\text{cost is \\textdollar{}5}", "escapes $ in \\text"); |
| 59 | eq(latexNormalizeForKatex("\\text{a & b % c_d ^ e ~ f}"), |
| 60 | "\\text{a \\& b \\% c\\_d \\textasciicircum{} e \\textasciitilde{} f}", |
| 61 | "escapes & % _ ^ ~ in \\text"); |
| 62 | eq(latexNormalizeForKatex("\\text{already\\_escaped}"), "\\text{already\\_escaped}", "no double-escape"); |
| 63 | eq(latexNormalizeForKatex("\\alpha + \\beta"), "\\alpha + \\beta", "non-text commands"); |
| 64 | eq(latexNormalizeForKatex("a | b"), "a \\vert b", "| to \\vert without doubled space"); |
| 65 | eq(latexNormalizeForKatex("|x|"), "\\vert x\\vert", "|x| keeps command boundary"); |
| 66 | eq(latexNormalizeForKatex("\\text{foo \\$ bar}"), "\\text{foo \\$ bar}", "already escaped $"); |
| 67 | eq(latexNormalizeForKatex("100%"), "100\\%", "raw % escaped to \\% (KaTeX comment-char fix)"); |
| 68 | eq(latexNormalizeForKatex("x = 50%"), "x = 50\\%", "% at end of math escaped"); |
| 69 | eq(latexNormalizeForKatex("a%b"), "a\\%b", "% between letters escaped"); |
| 70 | eq(latexNormalizeForKatex("a\\%b"), "a\\%b", "already-escaped \\% not double-escaped"); |
| 71 | eq(latexNormalizeForKatex("\\textrm{test #}"), "\\textrm{test \\#}", "\\textrm also handled"); |
| 72 | eq(latexNormalizeForKatex("\\textbf{hello world}"), "\\textbf{hello world}", "\\textbf no special chars"); |
| 73 | eq(latexNormalizeForKatex("\\tfrac{a}{b}"), "\\tfrac{a}{b}", "nested braces in command"); |
| 74 | eq(latexNormalizeForKatex("\\|x\\|"), "\\|x\\|", "\\| is left alone (readCommand handles \\|, not | branch)"); |
| 75 | eq(latexNormalizeForKatex("\\\\|x|"), "\\\\\\vert x\\vert", "\\\\| line break + pipe: both | → \\vert"); |
| 76 | |
| 77 | // ── latexNormalizeForKatex — array column-spec pipes (regression) ────────────── |
| 78 | // Inside \begin{array}{c|c} the | means "draw a vertical rule" — it must |
| 79 | // NOT be rewritten to \vert, or KaTeX fails with "Unknown column alignment: |
| 80 | // \vert". The whole {...} preamble is copied verbatim. |
| 81 | eq(latexNormalizeForKatex("\\begin{array}{c|c} a & b \\\\ c & d \\end{array}"), |
| 82 | "\\begin{array}{c|c} a & b \\\\ c & d \\end{array}", "array column-spec | preserved (c|c)"); |
| 83 | eq(latexNormalizeForKatex("\\begin{array}{|c|c|} a & b \\end{array}"), |
| 84 | "\\begin{array}{|c|c|} a & b \\end{array}", "array column-spec ||| preserved"); |
| 85 | eq(latexNormalizeForKatex("\\begin{array}{cc|c} a & b & c \\end{array}"), |
| 86 | "\\begin{array}{cc|c} a & b & c \\end{array}", "array column-spec cc|c preserved"); |
| 87 | eq(latexNormalizeForKatex("\\begin{array}{c|c} a & b \\end{array} |x|"), |
| 88 | "\\begin{array}{c|c} a & b \\end{array} \\vert x\\vert", "pipe OUTSIDE array still → \\vert"); |
| 89 | eq(latexNormalizeForKatex("\\begin{tabular}{c|c} a & b \\end{tabular}"), |
| 90 | "\\begin{tabular}{c|c} a & b \\end{tabular}", "tabular column-spec | preserved"); |
| 91 | |
| 92 | // ── latexNormalizeForKatex — ket-pipe disambiguation (regression) ───────────── |
| 93 | // In GFM Markdown tables, | is the column delimiter, so kets are written as |
| 94 | // \|uud\rangle. But \| is the "parallel-to" double bar ‖ in LaTeX, not a ket |
| 95 | // bar. We convert \| to \vert when it's a ket opener (\|...\rangle) or bra |
| 96 | // closer (\langle...\|), but leave matched \|...\| norms alone. |
| 97 | eq(latexNormalizeForKatex("\\|uud\\rangle"), "\\vert uud\\rangle", "ket \\|uud\\rangle → \\vert"); |
| 98 | eq(latexNormalizeForKatex("\\|\\alpha\\rangle"), "\\vert \\alpha\\rangle", "ket \\|\\alpha\\rangle → \\vert"); |
| 99 | eq(latexNormalizeForKatex("\\|u\\uparrow d\\rangle"), "\\vert u\\uparrow d\\rangle", "ket with content → \\vert"); |
| 100 | eq(latexNormalizeForKatex("\\frac{1}{\\sqrt{2}}\\|\\psi\\rangle"), "\\frac{1}{\\sqrt{2}}\\vert \\psi\\rangle", "ket in fraction → \\vert"); |
| 101 | eq(latexNormalizeForKatex("\\|a\\rangle + \\|b\\rangle"), "\\vert a\\rangle + \\vert b\\rangle", "two kets both → \\vert"); |
| 102 | // Norms (matched \|...\| pair) must KEEP the double bar |
| 103 | eq(latexNormalizeForKatex("\\|x\\|"), "\\|x\\|", "norm \\|x\\| preserved (double bar)"); |
| 104 | eq(latexNormalizeForKatex("\\|v\\|^2"), "\\|v\\|^2", "norm \\|v\\|^2 preserved"); |
| 105 | eq(latexNormalizeForKatex("\\|\\vec{v}\\|"), "\\|\\vec{v}\\|", "norm with content preserved"); |
| 106 | // Bra closers (\langle...\|) |
| 107 | eq(latexNormalizeForKatex("\\langle\\psi\\|"), "\\langle\\psi\\vert", "bra \\langle\\psi\\| → \\vert"); |
| 108 | // Inner product: \langle x \| y \rangle — the \| between bra and ket content |
| 109 | eq(latexNormalizeForKatex("\\langle x \\| y \\rangle"), "\\langle x \\vert y \\rangle", "inner product \\| → \\vert"); |
| 110 | |
| 111 | // ── latexNormalizeForKatex — \tag → align conversion (regression for KaTeX "Multiple \tag") ── |
| 112 | eq(latexNormalizeForKatex("a = b \\tag{10}"), "a = b \\tag{10}", "\\tag without aligned passes through"); |
| 113 | eq(latexNormalizeForKatex("\\begin{aligned} a &= b \\\\ \\end{aligned}"), |
| 114 | "\\begin{aligned} a &= b \\\\ \\end{aligned}", "aligned without \\tag unchanged"); |
| 115 | eq(latexNormalizeForKatex("\\begin{aligned} a &= b \\tag{10}\\\\ c &= d \\end{aligned}"), |
| 116 | "\\begin{align} a &= b \\tag{10}\\\\ c &= d \\end{align}", "aligned with \\tag → align"); |
| 117 | eq(latexNormalizeForKatex("\\begin{aligned} a &= b \\tag{10}\\\\ c &= d \\tag{11} \\end{aligned}"), |
| 118 | "\\begin{align} a &= b \\tag{10}\\\\ c &= d \\tag{11} \\end{align}", "aligned with multiple \\tag → align"); |
| 119 | eq(latexNormalizeForKatex("\\boxed{\\begin{aligned} a &= b \\tag{10}\\\\ c &= d \\end{aligned}}"), |
| 120 | "\\boxed{\\begin{align} a &= b \\tag{10}\\\\ c &= d \\end{align}}", "boxed aligned with \\tag → boxed align"); |
| 121 | eq(latexNormalizeForKatex("\\begin{gathered} a = b \\tag{10}\\\\ c = d \\end{gathered}"), |
| 122 | "\\begin{gather} a = b \\tag{10}\\\\ c = d \\end{gather}", "gathered with \\tag → gather"); |
| 123 | |
| 124 | // ── isLikelyInlineMath (mathClassify) ────────────────────────────────────────── |
| 125 | |
| 126 | console.log("\nisLikelyInlineMath — math"); |
| 127 | check("$x$ (single var)", () => isLikelyInlineMath("x") === true); |
| 128 | check("$E=mc^2$", () => isLikelyInlineMath("E=mc^2") === true); |
| 129 | check("$x_i^2$", () => isLikelyInlineMath("x_i^2") === true); |
| 130 | check("$\\alpha$", () => isLikelyInlineMath("\\alpha") === true); |
| 131 | check("$a \\le b$", () => isLikelyInlineMath("a \\le b") === true); |
| 132 | check("$\\frac{a}{b}$", () => isLikelyInlineMath("\\frac{a}{b}") === true); |
| 133 | check("$f(x)$", () => isLikelyInlineMath("f(x)") === true); |
| 134 | check("$x+1$", () => isLikelyInlineMath("x+1") === true); |
| 135 | |
| 136 | console.log("\nisLikelyInlineMath — classifier gaps from PR #4543"); |
| 137 | check("$\\tfrac12$", () => isLikelyInlineMath("\\tfrac12") === true); |
| 138 | check("$\\sqrt2$", () => isLikelyInlineMath("\\sqrt2") === true); |
| 139 | check("$SO(3,1)$", () => isLikelyInlineMath("SO(3,1)") === true); |
| 140 | check("$SU(2)$", () => isLikelyInlineMath("SU(2)") === true); |
| 141 | check("$GL(n)$", () => isLikelyInlineMath("GL(n)") === true); |
| 142 | check("$K = -iJ$", () => isLikelyInlineMath("K = -iJ") === true); |
| 143 | check("$p = +\\alpha$", () => isLikelyInlineMath("p = +\\alpha") === true); |
| 144 | check("$+$", () => isLikelyInlineMath("+") === true); |
| 145 | check("$=$", () => isLikelyInlineMath("=") === true); |
| 146 | |
| 147 | console.log("\nisLikelyInlineMath — numeric syntax"); |
| 148 | check("$5 is math by default (glued $ delimiters)", () => isLikelyInlineMath("5") === true); |
| 149 | check("$10 is math by default", () => isLikelyInlineMath("10") === true); |
| 150 | check("$10.50 is math by default", () => isLikelyInlineMath("10.50") === true); |
| 151 | check("$100% defaults to math", () => isLikelyInlineMath("100%") === true); |
| 152 | check("assistant-ui parity: price-word context no longer demotes pure numbers", () => |
| 153 | classifyInlineMath("5") === "math" && classifyInlineMath("10.50") === "math"); |
| 154 | check("range/unit/operator context is irrelevant now: pure numbers are always math", () => |
| 155 | classifyInlineMath("20") === "math" && classifyInlineMath("1") === "math"); |
| 156 | check("URL", () => isLikelyInlineMath("https://example.com") === false); |
| 157 | check("prose text", () => isLikelyInlineMath("hello world today") === false); |
| 158 | check("prose $x y z$ (spaces)", () => isLikelyInlineMath("x y z") === false); |
| 159 | check("$PATH$ env token", () => isLikelyInlineMath("PATH") === false); |
| 160 | check("$TODO$ word token", () => isLikelyInlineMath("TODO") === false); |
| 161 | check("$OK$ word token", () => isLikelyInlineMath("OK") === false); |
| 162 | check("$v1$ version token", () => isLikelyInlineMath("v1") === false); |
| 163 | check("$foo$ plain word", () => isLikelyInlineMath("foo") === false); |
| 164 | |
| 165 | console.log("\nisLikelyInlineMath — single-letter regression"); |
| 166 | check("lowercase $x$ → math", () => isLikelyInlineMath("x") === true); |
| 167 | check("uppercase $I$ → math (math name in non-English prose)", () => isLikelyInlineMath("I") === true); |
| 168 | check("uppercase $A$ → math", () => isLikelyInlineMath("A") === true); |
| 169 | check("uppercase $V$ → math", () => isLikelyInlineMath("V") === true); |
| 170 | |
| 171 | console.log("\nisLikelyInlineMath — primed letters and bracketed labels"); |
| 172 | check("$S'$ → math (primed letter)", () => isLikelyInlineMath("S'") === true); |
| 173 | check("$y''$ → math (double prime)", () => isLikelyInlineMath("y''") === true); |
| 174 | check("$f'(x)$ → math (primed function)", () => isLikelyInlineMath("f'(x)") === true); |
| 175 | check("$\\psi'$ → math (Greek with prime)", () => isLikelyInlineMath("\\psi'") === true); |
| 176 | check("$[56]$ → math (irrep label)", () => isLikelyInlineMath("[56]") === true); |
| 177 | check("$[56,0^+]$ → math", () => isLikelyInlineMath("[56,0^+]") === true); |
| 178 | check("$[\\mathbf{56}]$ → math", () => isLikelyInlineMath("[\\mathbf{56}]") === true); |
| 179 | |
| 180 | console.log("\nisLikelyInlineMath — minimal LaTeX patterns (regression)"); |
| 181 | // LLMs frequently emit minimal LaTeX in math contexts that the older |
| 182 | // classifier rejected as word tokens. These tests pin down the |
| 183 | // deliberately-permissive rules for common math patterns; pure numbers are |
| 184 | // math by default (assistant-ui parity). |
| 185 | check("single-digit $1$, $2$, $5$ → math by default", () => isLikelyInlineMath("1") === true); |
| 186 | check("multi-digit $42$ → math by default", () => isLikelyInlineMath("42") === true); |
| 187 | check("$2.5x$ is math (number with variable)", () => isLikelyInlineMath("2.5x") === true); |
| 188 | check("$10\%$ is math (percentage with LaTeX)", () => isLikelyInlineMath("10\\%") === true); |
| 189 | check("$2.5x dollars$ → NOT math (prefix-only numeric variable)", () => isLikelyInlineMath("2.5x dollars") === false); |
| 190 | check("$10\\% off$ → NOT math (prefix-only escaped percentage)", () => isLikelyInlineMath("10\\% off") === false); |
| 191 | check("$5\\cdot3$ is math (number with LaTeX command)", () => isLikelyInlineMath("5\\cdot3") === true); |
| 192 | |
| 193 | check("comma-separated $A, B$ → math (ordered pair)", () => isLikelyInlineMath("A, B") === true); |
| 194 | check("comma-separated $1, 2, 3$ → math (sequence)", () => isLikelyInlineMath("1, 2, 3") === true); |
| 195 | check("comma-separated $\\alpha, \\beta$ → math (Greek pair)", () => isLikelyInlineMath("\\alpha, \\beta") === true); |
| 196 | check("parens-wrapped $(A, B)$ inner → math", () => isLikelyInlineMath("(A, B)") === true); |
| 197 | check("cycle notation $(12)$ → math", () => isLikelyInlineMath("(12)") === true); |
| 198 | check("cycle notation $(12)(34)$ → math", () => isLikelyInlineMath("(12)(34)") === true); |
| 199 | check("$S$ (set name) → math", () => isLikelyInlineMath("S") === true); |
| 200 | check("$S$ with surrounding prose (regression)", () => { |
| 201 | return normalizeMath("$S$ 非空\n$S$ 有上界") === "$S$ 非空\n$S$ 有上界"; |
| 202 | }); |
| 203 | check("one-sided comparison $< B$ → math", () => isLikelyInlineMath("< B") === true); |
| 204 | check("one-sided comparison $<= 0$ → math", () => isLikelyInlineMath("<= 0") === true); |
| 205 | check("one-sided comparison $> 5$ → math", () => isLikelyInlineMath("> 5") === true); |
| 206 | check("one-sided comparison $A <$ → math", () => isLikelyInlineMath("A <") === true); |
| 207 | check("one-sided equality $=1$ → math", () => isLikelyInlineMath("=1") === true); |
| 208 | check("one-sided signed equality $=-1$ → math", () => isLikelyInlineMath("=-1") === true); |
| 209 | check("one-sided equality is fully anchored", () => isLikelyInlineMath("=1 dollar") === false); |
| 210 | check("$< B$ with surrounding prose", () => { |
| 211 | return normalizeMath("A 的每个元素 $< B$ 的每个元素") === "A 的每个元素 $< B$ 的每个元素"; |
| 212 | }); |
| 213 | |
| 214 | // ── KaTeX end-to-end rendering ──────────────────────────────────────────────── |
| 215 | |
| 216 | const chiralSource = String.raw` |
| 217 | \underbrace{N}_{\text{baryon #}} |
| 218 | = |
| 219 | \underbrace{\frac{1+\tau_3}{2}}_{\text{isospin}} |
| 220 | + |
| 221 | \underbrace{g_A \gamma^\mu \gamma_5}_{\text{axial}} |
| 222 | + |
| 223 | \underbrace{SU(2)_L \times SU(2)_R}_{\text{chiral}} |
| 224 | `; |
| 225 | |
| 226 | function renderDisplay(source: string): string { |
| 227 | return katex.renderToString(latexNormalizeForKatex(source), { |
| 228 | throwOnError: true, |
| 229 | displayMode: true, |
| 230 | }); |
| 231 | } |
| 232 | |
| 233 | console.log("\nKaTeX renderToString — end to end"); |
| 234 | check("chiral decomposition renders", () => { |
| 235 | const html = renderDisplay(chiralSource); |
| 236 | return !html.includes("katex-error") |
| 237 | && ["baryon", "isospin", "axial", "chiral"].every((label) => html.includes(label)); |
| 238 | }); |
| 239 | check("\\|x\\| renders as double bars", () => { |
| 240 | const html = renderDisplay(String.raw`\|x\|`); |
| 241 | return !html.includes("katex-error") && html.includes("∥"); |
| 242 | }); |
| 243 | |
| 244 | // ── normalizeMath pre-pass (LLM delimiters + classifier) ─────────────────────── |
| 245 | // These exercise the *production* normalizeMath, not a copy of it. |
| 246 | |
| 247 | console.log("\nnormalizeMath — LLM delimiter conversion"); |
| 248 | eq(normalizeMath("\\(x^2\\)"), "$x^2$", "\\(…\\) → $…$"); |
| 249 | eq(normalizeMath("\\[E=mc^2\\]"), "$$\nE=mc^2\n$$", "\\[…\\] → $$…$$"); |
| 250 | eq(normalizeMath("\\\\[4pt]"), "\\\\[4pt]", "\\\\[ line-break spacing protected"); |
| 251 | |
| 252 | console.log("\nnormalizeMath — \\slashed conversion (regression)"); |
| 253 | // KaTeX has no \slashed (Feynman slash notation). The pre-pass preserves it |
| 254 | // verbatim; the AST policy rewrites it to \not only for rendering. |
| 255 | eq(normalizeMath("$\\slashed{p}$"), "$\\slashed{p}$", "\\slashed{p} deferred to AST policy"); |
| 256 | eq(normalizeMath("$\\slashed{\\partial}$"), "$\\slashed{\\partial}$", "\\slashed{\\partial} deferred to AST policy"); |
| 257 | eq( |
| 258 | normalizeMath("The momentum $\\slashed{p}$ is conserved"), |
| 259 | "The momentum $\\slashed{p}$ is conserved", |
| 260 | "\\slashed in prose deferred to AST policy", |
| 261 | ); |
| 262 | eq(normalizeMath("$\\slashed\\epsilon(0)$"), "$\\slashed\\epsilon(0)$", "unbraced \\slashed normalisation deferred to AST policy"); |
| 263 | eq(normalizeMath("$\\slashed a$"), "$\\slashed a$", "unbraced \\slashed letter normalisation deferred to AST policy"); |
| 264 | |
| 265 | console.log("\nnormalizeMath — inline $$ glued to prose (regression)"); |
| 266 | // User-reported: "…decomposes as$$\n\mathbf{6}…" — block math glued to prose. |
| 267 | // Without a blank line, remark-math parses the opening $$ as an empty math node |
| 268 | // and the formula leaks out as literal text. normalizeMath must insert a blank |
| 269 | // line before any $$ preceded by a letter/closing bracket/etc. |
| 270 | check("inline $$ after prose", () => { |
| 271 | const out = normalizeMath("decomposes as$$\n\\mathbf{6}.$$"); |
| 272 | return /^decomposes as\n\$\$/.test(out) && out.includes("\\mathbf{6}"); |
| 273 | }); |
| 274 | check("inline $$ after closing bracket", () => { |
| 275 | const out = normalizeMath("(octet)$$ \\mathbf{56}.$$"); |
| 276 | return out.startsWith("(octet)\n$$"); |
| 277 | }); |
| 278 | check("inline $$ after closing brace (\\end{...}$$)", () => { |
| 279 | // A display equation ending with }$$ must be extracted as a unit. |
| 280 | // The closing $$ must not be split off, or the equation body is emptied. |
| 281 | const out = normalizeMath("$$\\begin{pmatrix}a&b\\\\c&d\\end{pmatrix}$$"); |
| 282 | return out.includes("\\end{pmatrix},\n$$") || out.includes("\\end{pmatrix}\n$$"); |
| 283 | }); |
| 284 | check("inline $$ after comma on same line as content", () => { |
| 285 | // User-reported (2026-06-12, soft-pion chat): the model wrote the |
| 286 | // closing $$ of a display block on the same line as the trailing |
| 287 | // comma of the equation content, like |
| 288 | // …D(q^2),$$ |
| 289 | // with $P=…$ |
| 290 | // Without a blank line before the closing $$, micromark-extension-math |
| 291 | // does not recognise the closing fence (it only checks for $$ at |
| 292 | // the start of a new line) and consumes the rest of the document |
| 293 | // as math, which then fails to render with "Can't use function '$' |
| 294 | // in math mode" on the stray $ inside the equation body. |
| 295 | const out = normalizeMath("…D(q^2),$$\nwith $P=…$"); |
| 296 | return out.includes("D(q^2),\n$$"); |
| 297 | }); |
| 298 | check("well-formed $$ already on own line is normalised consistently", () => { |
| 299 | // Whether the model writes `decomposes as$$\n\mathbf{6}.$$` or |
| 300 | // `decomposes as\n\n$$\n\mathbf{6}.$$`, both must produce valid |
| 301 | // remark-math-parseable form: opening $$ on its own line, body, closing |
| 302 | // $$ on its own line. |
| 303 | const inline = normalizeMath("decomposes as$$\n\\mathbf{6}.$$"); |
| 304 | const block = normalizeMath("decomposes as\n\n$$\n\\mathbf{6}.$$"); |
| 305 | const valid = (s: string) => /\n\$\$\n/.test(s) && /\n\$\$/.test(s) && s.includes("\\mathbf{6}"); |
| 306 | return valid(inline) && valid(block); |
| 307 | }); |
| 308 | check("\\[…\\] → $$…$$ still works (no spurious blank line)", () => { |
| 309 | return normalizeMath("\\[E=mc^2\\]") === "$$\nE=mc^2\n$$"; |
| 310 | }); |
| 311 | check("digit before $$ is NOT a prose boundary (preserves c^2$$)", () => { |
| 312 | const out = normalizeMath("c^2$$ x $$"); |
| 313 | return out === "c^2$$ x $$"; |
| 314 | }); |
| 315 | eq(normalizeMath("intro$$x+1"), "intro\n$$\nx+1", "orphan opening $$ is not duplicated"); |
| 316 | eq( |
| 317 | normalizeMath("first$$a$$ middle $$b$$ end"), |
| 318 | "first\n$$\na\n$$\n middle \n$$\nb\n$$\n end", |
| 319 | "multiple display blocks on one line are all normalised", |
| 320 | ); |
| 321 | |
| 322 | console.log("\nnormalizeMath — semantic dollar decisions deferred to AST policy"); |
| 323 | eq(normalizeMath("costs $1$ today"), "costs $1$ today", "$1$ preserved for contextual classification"); |
| 324 | eq(normalizeMath("env $PATH$ here"), "env $PATH$ here", "$PATH$ preserved for AST literal restoration"); |
| 325 | eq(normalizeMath("solve $x^2 + y^2 = z^2$ please"), "solve $x^2 + y^2 = z^2$ please", "$x^2+y^2$ is math"); |
| 326 | eq(normalizeMath("$\\alpha + \\beta$"), "$\\alpha + \\beta$", "$\\alpha+\\beta$ is math"); |
| 327 | eq(normalizeMath("price is $10.50$ each"), "price is $10.50$ each", "$10.50$ preserved for contextual classification"); |
| 328 | eq(normalizeMath("$I$ think"), "$I$ think", "$I$ is math (uppercase single letter)"); |
| 329 | eq(normalizeMath("it costs $5 and $10 total"), "it costs \\$5 and \\$10 total", "cross-amount prose dollars escaped by the currency pre-pass"); |
| 330 | |
| 331 | console.log("\nnormalizeMath — Markdown code regions stay literal"); |
| 332 | eq(normalizeMath("`$PATH$`"), "`$PATH$`", "inline code with env token"); |
| 333 | eq(normalizeMath("Use `$HOME` and `$PATH$`."), "Use `$HOME` and `$PATH$`.", "multiple inline code spans"); |
| 334 | eq(normalizeMath("```sh\necho $PATH$\n```"), "```sh\necho $PATH$\n```", "fenced code with env token"); |
| 335 | eq(normalizeMath("```\necho $PATH$\n```\n\nsolve $x^2$"), "```\necho $PATH$\n```\n\nsolve $x^2$", "fenced code protected while prose math renders"); |
| 336 | eq(normalizeMath("Code: `r.replace(/\\$\\$/, ...)`"), "Code: `r.replace(/\\$\\$/, ...)`", "escaped $ in inline code stays literal"); |
| 337 | eq(normalizeMath("```javascript\nr = r.replace(/\\$\\$([\\s\\S]*?)\\$\\$/g, ...);\n```"), "```javascript\nr = r.replace(/\\$\\$([\\s\\S]*?)\\$\\$/g, ...);\n```", "regex patterns with $ in code blocks stay literal"); |
| 338 | eq(normalizeMath("Code: `` `${DOLLAR}${m}${DOLLAR}` ``"), "Code: `` `${DOLLAR}${m}${DOLLAR}` ``", "template literals with $ in inline code stay literal"); |
| 339 | |
| 340 | // ── normalizeMath — text-mode source protection (regression for PR #3287) ───── |
| 341 | // A stray inner $ must be hidden until remark-math establishes the outer |
| 342 | // boundary, while the AST policy retains the exact source for copying. |
| 343 | |
| 344 | console.log("\nnormalizeMath — text-mode escapes (regression)"); |
| 345 | check("$\\text{cost is $5}$ inner $ is parser-safe and reversible", () => { |
| 346 | const out = normalizeMath("$\\text{cost is $5}$"); |
| 347 | return !out.slice(1, -1).includes("$") |
| 348 | && restoreProtectedInlineMathSource(out.slice(1, -1)) === "\\text{cost is $5}"; |
| 349 | }); |
| 350 | check("$\\text{baryon #}$ # escape is deferred to AST policy", () => { |
| 351 | return normalizeMath("$\\text{baryon #}$") === "$\\text{baryon #}$"; |
| 352 | }); |
| 353 | check("$\\text{a & b}$ & escape is deferred to AST policy", () => { |
| 354 | return normalizeMath("$\\text{a & b}$") === "$\\text{a & b}$"; |
| 355 | }); |
| 356 | check("$\\text{cost is \\$5}$ escaped dollar stays literal", () => { |
| 357 | return normalizeMath("$\\text{cost is \\$5}$") === "$\\text{cost is \\$5}$"; |
| 358 | }); |
| 359 | check("$\\textrm{cost is \\$5}$ escaped dollar stays literal", () => { |
| 360 | return normalizeMath("$\\textrm{cost is \\$5}$") === "$\\textrm{cost is \\$5}$"; |
| 361 | }); |
| 362 | check("$\\sqrt{x}$ non-text command preserved", () => { |
| 363 | return normalizeMath("$\\sqrt{x}$") === "$\\sqrt{x}$"; |
| 364 | }); |
| 365 | |
| 366 | // ── normalizeMath — TEXT_MODE_PAIR trailing content ────────────────────────────── |
| 367 | // $\cmd{...} + extra$ should be handled as a whole, not split at inner $. |
| 368 | |
| 369 | console.log("\nnormalizeMath — TEXT_MODE_PAIR trailing content"); |
| 370 | check("$\\text{cost is $5} + x^2$ inner $ escaped with trailing", () => { |
| 371 | const out = normalizeMath("$\\text{cost is $5} + x^2$"); |
| 372 | return restoreProtectedInlineMathSource(out.slice(1, -1)) |
| 373 | === "\\text{cost is $5} + x^2"; |
| 374 | }); |
| 375 | check("$\\text{a} | b$ pipe after text command", () => { |
| 376 | const out = normalizeMath("$\\text{a} | b$"); |
| 377 | return restoreProtectedInlineMathSource(out.slice(1, -1)) === "\\text{a} | b"; |
| 378 | }); |
| 379 | check("$\\text{abc}$ simple text-mode (no trailing)", () => { |
| 380 | return normalizeMath("$\\text{abc}$") === "$\\text{abc}$"; |
| 381 | }); |
| 382 | |
| 383 | // ── normalizeMath — GFM pipe protection (raw | marked, \\| preserved) ────────── |
| 384 | |
| 385 | console.log("\nnormalizeMath — pipe handling"); |
| 386 | check("$|x+1|$ absolute value", () => { |
| 387 | const out = normalizeMath("$|x+1|$"); |
| 388 | return restoreProtectedInlineMathSource(out.slice(1, -1)) === "|x+1|"; |
| 389 | }); |
| 390 | check("$\\|x\\|$ norm preserved (no \\vert mangling)", () => { |
| 391 | return normalizeMath("$\\|x\\|$") === "$\\|x\\|$"; |
| 392 | }); |
| 393 | |
| 394 | // ── normalizeMath — % in math (KaTeX comment-char) ───────────────────────────── |
| 395 | // KaTeX treats unescaped % as a LaTeX comment to end-of-line, silently |
| 396 | // truncating `$x = 50%$` to `$x = 50$`. Top-level % must be escaped. |
| 397 | |
| 398 | console.log("\nnormalizeMath — % in math"); |
| 399 | eq(normalizeMath("$x = 50%$"), "$x = 50%$", "trailing % escape deferred to AST policy"); |
| 400 | eq(normalizeMath("$100%$"), "$100%$", "pure percentage preserved for AST math policy"); |
| 401 | eq(normalizeMath("$10\\%$"), "$10\\%$", "already-escaped \\% left alone"); |
| 402 | |
| 403 | // ── normalizer + AST policy — end-to-end KaTeX render of common LLM outputs ─── |
| 404 | |
| 405 | console.log("\nnormalizeMath → KaTeX end-to-end"); |
| 406 | function katexOf(normalized: string, display: boolean): boolean { |
| 407 | let inner: string; |
| 408 | if (normalized.startsWith("$$") && normalized.endsWith("$$")) { |
| 409 | inner = normalized.slice(2, -2); |
| 410 | display = true; |
| 411 | } else if (normalized.startsWith("$") && normalized.endsWith("$")) { |
| 412 | inner = normalized.slice(1, -1); |
| 413 | } else { |
| 414 | return false; // no math delimiters — nothing for KaTeX to render |
| 415 | } |
| 416 | try { |
| 417 | const source = restoreProtectedInlineMathSource(inner); |
| 418 | katex.renderToString(latexNormalizeForKatex(source), { |
| 419 | throwOnError: true, |
| 420 | displayMode: display, |
| 421 | }); |
| 422 | return true; |
| 423 | } catch { |
| 424 | return false; |
| 425 | } |
| 426 | } |
| 427 | |
| 428 | const e2e: Array<[string, string]> = [ |
| 429 | ["$\\text{cost is $5}$", "text mode with literal $"], |
| 430 | ["$\\text{baryon #}$", "text mode with #"], |
| 431 | ["$\\text{a & b}$", "text mode with &"], |
| 432 | ["$\\|x\\|$", "norm"], |
| 433 | ["$|x+1|$", "abs value"], |
| 434 | ["$x=1$", "simple equation"], |
| 435 | ["$\\frac{a}{b}$", "fraction"], |
| 436 | ["$\\alpha + \\beta$", "greek letters"], |
| 437 | ["$ \\sqrt{x} $", "sqrt with surrounding spaces"], |
| 438 | ["$$E=mc^2$$", "display equation"], |
| 439 | ["\\(\\alpha\\)", "LLM-native inline delimiter"], |
| 440 | ["\\[\\sum_{i=1}^n i\\]", "LLM-native display delimiter"], |
| 441 | ["$$ |a| = |b| $$", "display with absolute values"], |
| 442 | ["$$\\boxed{\\begin{aligned}\nr_A E_\\pi(k;0) &= B(k^2) \\\\\nF_R(k;0) + 2r_A F_\\pi(k;0) &= A(k^2)\n\\end{aligned}}$$", "boxed aligned (no \\tag)"], |
| 443 | ["$$\\boxed{\\begin{aligned}\nr_A E_\\pi(k;0) &= B(k^2) \\tag{10}\\\\\nF_R(k;0) + 2r_A F_\\pi(k;0) &= A(k^2) \\tag{11}\n\\end{aligned}}$$", "boxed aligned with \\tag → align (no error)"], |
| 444 | ["\\[\\boxed{\\begin{aligned}\nx &= 1 \\\\\ny &= 2\n\\end{aligned}}\\]", "LLM-native boxed aligned"], |
| 445 | // Array with column-spec pipe — regression: |→\vert used to corrupt {c|c} |
| 446 | // into {c\vert c} (KaTeX: "Unknown column alignment"). Must render cleanly. |
| 447 | ["$$\\begin{array}{c|c} a & b \\\\ c & d \\end{array}$$", "array with c|c column spec"], |
| 448 | ["$$\\begin{array}{cc|c} a & b & c \\\\ d & e & f \\end{array}$$", "array with cc|c column spec"], |
| 449 | ["$$\\begin{array}{|c|c|} a & b \\\\ c & d \\end{array}$$", "array with |c|c| column spec"], |
| 450 | // Ket with \| delimiter (common in GFM tables where | must be escaped) |
| 451 | ["$\\|\\psi\\rangle$", "ket with \\| → single bar (regression)"], |
| 452 | ["$\\frac{1}{\\sqrt{2}}\\|uud\\rangle$", "ket in fraction with \\|"], |
| 453 | ["$\\|x\\|$", "norm \\|x\\| → double bar (regression)"], |
| 454 | ["$\\langle\\psi\\|$", "bra closer \\| → single bar (regression)"], |
| 455 | ["$S'$", "primed letter S'"], |
| 456 | ["$f'(x)$", "primed function call"], |
| 457 | ["$[56]$", "bracketed irrep label"], |
| 458 | ["$[56,0^+]$", "bracketed irrep label with charge"], |
| 459 | ]; |
| 460 | for (const [src, label] of e2e) { |
| 461 | check(`${label}: ${src}`, () => katexOf(normalizeMath(src), false)); |
| 462 | } |
| 463 | |
| 464 | // Inputs that contain no math delimiters must survive normalizeMath |
| 465 | // unchanged — KaTeX isn't involved here. |
| 466 | console.log("\nnormalizeMath — non-math inputs pass through"); |
| 467 | type Passthrough = { src: string; expected: string; label: string }; |
| 468 | const passthrough: Passthrough[] = [ |
| 469 | { src: "costs $100$ today", expected: "costs $100$ today", label: "multi-digit pair passes through the pre-pass untouched" }, |
| 470 | { src: "line break \\\\[4pt] here", expected: "line break \\\\[4pt] here", label: "LaTeX line-break spacing" }, |
| 471 | { src: "hello world", expected: "hello world", label: "plain text" }, |
| 472 | ]; |
| 473 | for (const { src, expected, label } of passthrough) { |
| 474 | check(`${label}: ${src}`, () => normalizeMath(src) === expected); |
| 475 | } |
| 476 | |
| 477 | // ── remark-math render boundary ──────────────────────────────────────────────── |
| 478 | // These cases cross the real react-markdown → remark-math → Reasonix AST |
| 479 | // policy → rehype-katex boundary. The policy restores literal nodes after |
| 480 | // parsing, so rejected content cannot be reparsed as math. |
| 481 | |
| 482 | console.log("\nnormalizeMath → remark-math render boundary"); |
| 483 | |
| 484 | function renderHtml(src: string): string { |
| 485 | return renderToStaticMarkup( |
| 486 | createElement(ReactMarkdown, { |
| 487 | remarkPlugins: reasonixRemarkPlugins, |
| 488 | rehypePlugins: reasonixRehypePlugins, |
| 489 | children: normalizeMath(src), |
| 490 | }), |
| 491 | ); |
| 492 | } |
| 493 | |
| 494 | check("cross-amount pairing '$5 and $6' renders as literal dollars, not math", () => { |
| 495 | const html = renderHtml("These two apples cost $5 and $6"); |
| 496 | return !html.includes("katex") && html.includes("$5") && html.includes("$6"); |
| 497 | }); |
| 498 | // Assistant-ui `escapeCurrencyDollars` parity: a glued $N$ pair is math even |
| 499 | // when price words or currency units sit next to it. Humans and models that |
| 500 | // want literal dollars write `\$5` (escaped) or a single `$5`; the cross- |
| 501 | // amount prose pair above is the real-world currency artifact, and it stays |
| 502 | // literal via the classifier catch-all rather than this demotion. |
| 503 | const PARITY_MATH_CASES: ReadonlyArray<readonly [string, string]> = [ |
| 504 | ["costs $1$ today", "1"], |
| 505 | ["price is $10.50$ each", "10.50"], |
| 506 | ["价格是$5$", "5"], |
| 507 | ["价格:$5$", "5"], |
| 508 | ["The price ($5$) includes tax.", "5"], |
| 509 | ["The price {$5$} includes tax.", "5"], |
| 510 | ["The price ‘$5$’ includes tax.", "5"], |
| 511 | ["价格($5$)含税。", "5"], |
| 512 | ["It is $5$ (USD).", "5"], |
| 513 | ["It is $5$—cash.", "5"], |
| 514 | ["I have $5$ in cash", "5"], |
| 515 | ["costs **$5$** today", "5"], |
| 516 | ["price is *$10.50$* each", "10.50"], |
| 517 | ]; |
| 518 | for (const [src, num] of PARITY_MATH_CASES) { |
| 519 | check(`assistant-ui parity: "${src}" renders as math`, () => { |
| 520 | const html = renderHtml(src); |
| 521 | return html.includes("katex") && html.includes(`<mn>${num}</mn>`); |
| 522 | }); |
| 523 | } |
| 524 | check("escaped \\$ dollars stay literal and never pair", () => { |
| 525 | const html = renderHtml("It costs \\$5 today, not \\$6"); |
| 526 | return !html.includes("katex") && html.includes("$5") && html.includes("$6"); |
| 527 | }); |
| 528 | // Currency pre-pass (assistant-ui escapeCurrencyDollars parity): a `$` |
| 529 | // followed by a digit is escaped unless its span to the next `$` reads as a |
| 530 | // math body, so a stray amount can no longer swallow a later formula. |
| 531 | check("lone unpaired $5 stays literal", () => { |
| 532 | const html = renderHtml("It costs $5 today."); |
| 533 | return !html.includes("katex") && html.includes("$5 today."); |
| 534 | }); |
| 535 | check("currency $ escapes so a later math span renders: 'budget is $100 … $42$'", () => { |
| 536 | const html = renderHtml("The budget is $100 and the answer is $42$."); |
| 537 | return html.includes("katex") && html.includes("<mn>42</mn>") |
| 538 | && html.includes("$100") && !html.includes("$42$"); |
| 539 | }); |
| 540 | check("currency $ escapes so a later symbol formula renders", () => { |
| 541 | const html = renderHtml("It costs $5, and $x+y$ is the formula."); |
| 542 | return html.includes("katex") && !html.includes("$x+y$"); |
| 543 | }); |
| 544 | check("digit-led math bodies survive currency escaping", () => { |
| 545 | const html = renderHtml("digit-led $2x$ and $5x = 10$ both render"); |
| 546 | return html.includes("katex") && !html.includes("$2x$") && !html.includes("$5x = 10$"); |
| 547 | }); |
| 548 | eq(normalizeMath("The budget is $100 and the answer is $42$."), |
| 549 | "The budget is \\$100 and the answer is $42$.", |
| 550 | "currency pre-pass escapes only the amount dollar"); |
| 551 | check("paired numbers 'from $5$ to $10$' render as math (global default)", () => { |
| 552 | const html = renderHtml("from $5$ to $10$"); |
| 553 | return html.includes("katex") && html.includes("<mn>5</mn>") && html.includes("<mn>10</mn>") |
| 554 | && !html.includes("$5$") && !html.includes("$10$"); |
| 555 | }); |
| 556 | check("env var $PATH$ renders as literal, not math", () => { |
| 557 | const html = renderHtml("env $PATH$ here"); |
| 558 | return !html.includes("katex") && html.includes("$PATH$"); |
| 559 | }); |
| 560 | check("range endpoint 10–$20$ MeV renders numeric math", () => { |
| 561 | const html = renderHtml("10–$20$ MeV"); |
| 562 | return html.includes("katex") && html.includes("<mn>20</mn>"); |
| 563 | }); |
| 564 | check("standalone $42$ renders as math without other context", () => { |
| 565 | const html = renderHtml("$42$ elements"); |
| 566 | return html.includes("katex") && html.includes("<mn>42</mn>") && !html.includes("$42$"); |
| 567 | }); |
| 568 | // GFM table cells: pure numbers render as math by default (GitHub and |
| 569 | // assistant-ui parity), in cells and prose alike. |
| 570 | check("pure number $1$ in a GFM table cell renders as math", () => { |
| 571 | const html = renderHtml([ |
| 572 | "| Quantity | SU(6) prediction | Experiment |", |
| 573 | "|---|---|---|", |
| 574 | "| $\\Delta\\Sigma$ | $1$ | $1.2754$ |", |
| 575 | ].join("\n")); |
| 576 | return html.includes("katex") && html.includes("<mn>1</mn>") && html.includes("<mn>1.2754</mn>") |
| 577 | && !html.includes("$1$") && !html.includes("$1.2754$"); |
| 578 | }); |
| 579 | check("bold pure number in a GFM table cell still renders as math", () => { |
| 580 | const html = renderHtml(["| a | b |", "|---|---|", "| **$5$** | $3$ |"].join("\n")); |
| 581 | return html.includes("katex") && html.includes("<mn>5</mn>") && html.includes("<mn>3</mn>") |
| 582 | && !html.includes("$5$") && !html.includes("$3$"); |
| 583 | }); |
| 584 | check("currency prose inside a GFM table cell renders as math (parity)", () => { |
| 585 | const html = renderHtml(["| note |", "|---|", "| costs $5$ today |"].join("\n")); |
| 586 | return html.includes("katex") && html.includes("<mn>5</mn>"); |
| 587 | }); |
| 588 | // Spacing & pairing: remark-math v6 pairs $…$ greedily, even across spaces |
| 589 | // and words, so these judgements live in the policy, not the tokenizer. |
| 590 | // Content is trimmed before classification, and cross-word pairs restore |
| 591 | // verbatim because they match no math pattern. |
| 592 | check("sloppy spaced delimiters with price words still render as math (parity)", () => { |
| 593 | const html = renderHtml("It costs $ 5$ today"); |
| 594 | return html.includes("katex") && html.includes("<mn>5</mn>"); |
| 595 | }); |
| 596 | check("spaced closing delimiter with cash context stays literal (assistant-ui parity)", () => { |
| 597 | const html = renderHtml("I paid $5 $ cash"); |
| 598 | return !html.includes("katex") && html.includes("$5") && !html.includes("$5$"); |
| 599 | }); |
| 600 | check("cross-word $…$ pairing restores verbatim, never math", () => { |
| 601 | const html = renderHtml("from $5 to $10"); |
| 602 | return !html.includes("katex") && html.includes("from $5 to $10"); |
| 603 | }); |
| 604 | check("bare number with sloppy spaced delimiters renders as math (global default)", () => { |
| 605 | const html = renderHtml("the value is $ 5$ here"); |
| 606 | return html.includes("katex") && html.includes("<mn>5</mn>"); |
| 607 | }); |
| 608 | check("scientific unit makes a paired number mathematical", () => { |
| 609 | const html = renderHtml("$20$ MeV"); |
| 610 | return html.includes("katex") && html.includes("<mn>20</mn>"); |
| 611 | }); |
| 612 | check("centimetres make a paired number mathematical", () => { |
| 613 | const html = renderHtml("$5$ cm"); |
| 614 | return html.includes("katex") && html.includes("<mn>5</mn>"); |
| 615 | }); |
| 616 | check("litres make a paired number mathematical", () => { |
| 617 | const html = renderHtml("$2$ L"); |
| 618 | return html.includes("katex") && html.includes("<mn>2</mn>"); |
| 619 | }); |
| 620 | check("decibels make a paired number mathematical", () => { |
| 621 | const html = renderHtml("$3$ dB"); |
| 622 | return html.includes("katex") && html.includes("<mn>3</mn>"); |
| 623 | }); |
| 624 | check("parenthesized scientific quantity remains mathematical", () => { |
| 625 | const html = renderHtml("A vector ($5$ m) long."); |
| 626 | return html.includes("katex") && html.includes("<mn>5</mn>"); |
| 627 | }); |
| 628 | check("parenthesized ambiguous value retains mathematical wrapper context", () => { |
| 629 | const html = renderHtml("The value ($5$) is exact."); |
| 630 | return html.includes("katex") && html.includes("<mn>5</mn>"); |
| 631 | }); |
| 632 | check("one-sided equality $=1$ renders as math", () => { |
| 633 | const html = renderHtml("set $=1$ here"); |
| 634 | return html.includes("katex") && html.includes("<mo>=</mo>"); |
| 635 | }); |
| 636 | check("one-sided equality does not prefix-match prose", () => { |
| 637 | const html = renderHtml("set $=1 dollar$ here"); |
| 638 | return !html.includes("katex") && html.includes("$=1 dollar$"); |
| 639 | }); |
| 640 | check("inline code remains outside math policy", () => { |
| 641 | const html = renderHtml("code `$42$` and env `$PATH$`"); |
| 642 | return !html.includes("katex") && html.includes("<code>$42$</code>") && html.includes("<code>$PATH$</code>"); |
| 643 | }); |
| 644 | check("AST policy applies KaTeX percent normalisation", () => { |
| 645 | const html = renderHtml("$x = 50%$"); |
| 646 | return html.includes("katex") |
| 647 | && !html.includes("katex-error") |
| 648 | && html.includes('data-latex-source="x = 50%"') |
| 649 | && html.includes('encoding="application/x-tex">x = 50%</annotation>'); |
| 650 | }); |
| 651 | check("AST policy applies unbraced slashed normalisation", () => { |
| 652 | const html = renderHtml("$\\slashed a$"); |
| 653 | return html.includes("katex") |
| 654 | && !html.includes("katex-error") |
| 655 | && html.includes('data-latex-source="\\slashed a"') |
| 656 | && html.includes('encoding="application/x-tex">\\slashed a</annotation>'); |
| 657 | }); |
| 658 | check("AST policy preserves braced slashed source after rendering", () => { |
| 659 | const html = renderHtml("$\\slashed{p}$"); |
| 660 | return html.includes("katex") |
| 661 | && !html.includes("katex-error") |
| 662 | && html.includes('data-latex-source="\\slashed{p}"') |
| 663 | && html.includes('encoding="application/x-tex">\\slashed{p}</annotation>'); |
| 664 | }); |
| 665 | check("parser-safe text-mode math restores the exact copy source", () => { |
| 666 | const html = renderHtml("$\\text{cost is $5}$"); |
| 667 | return html.includes("katex") |
| 668 | && !html.includes("katex-error") |
| 669 | && html.includes('data-latex-source="\\text{cost is $5}"') |
| 670 | && html.includes('encoding="application/x-tex">\\text{cost is $5}</annotation>'); |
| 671 | }); |
| 672 | check("real inline math $x^2$ still renders as KaTeX", () => { |
| 673 | const html = renderHtml("the value $x^2$ here"); |
| 674 | return html.includes("katex"); |
| 675 | }); |
| 676 | check("inline math with asymmetric delimiter padding still renders", () => { |
| 677 | const html = renderHtml("before $\\alpha $ after"); |
| 678 | return html.includes("katex") |
| 679 | && html.includes('data-latex-source="\\alpha "') |
| 680 | && html.includes('encoding="application/x-tex">\\alpha </annotation>'); |
| 681 | }); |
| 682 | check("inline math with multiple delimiter spaces still renders", () => { |
| 683 | const html = renderHtml("before $ x $ after"); |
| 684 | return html.includes("katex") |
| 685 | && html.includes('data-latex-source=" x "') |
| 686 | && html.includes('encoding="application/x-tex"> x </annotation>'); |
| 687 | }); |
| 688 | check("GFM table preserves inline absolute-value math and every cell", () => { |
| 689 | const html = renderHtml("| Expr | Value |\n| --- | --- |\n| $|x|$ | abs |"); |
| 690 | return html.includes("<table>") |
| 691 | && html.includes("katex") |
| 692 | && html.includes('data-latex-source="|x|"') |
| 693 | && html.includes('encoding="application/x-tex">|x|</annotation>') |
| 694 | && html.includes("<td>abs</td>") |
| 695 | && (html.match(/<td>/g) ?? []).length === 2; |
| 696 | }); |
| 697 | check("display math preserves original TeX in the KaTeX root and annotation", () => { |
| 698 | const html = renderHtml("$$|x|$$"); |
| 699 | return html.includes('class="katex-display" data-latex-source="|x|"') |
| 700 | && html.includes('encoding="application/x-tex">|x|</annotation>'); |
| 701 | }); |
| 702 | check("inline Young diagrams preserve their authored macro source", () => { |
| 703 | const html = renderHtml("before $V=\\yng(2,1)$ after"); |
| 704 | return html.includes("katex") |
| 705 | && !html.includes("katex-error") |
| 706 | && !html.includes("reasonixInternal") |
| 707 | && html.includes('data-latex-source="V=\\yng(2,1)"') |
| 708 | && html.includes('encoding="application/x-tex">V=\\yng(2,1)</annotation>'); |
| 709 | }); |
| 710 | check("display Young tableaux preserve their authored macro source", () => { |
| 711 | const html = renderHtml("$$\\young(ab,c)$$"); |
| 712 | return html.includes("katex-display") |
| 713 | && !html.includes("katex-error") |
| 714 | && !html.includes("reasonixInternal") |
| 715 | && html.includes('data-latex-source="\\young(ab,c)"') |
| 716 | && html.includes('encoding="application/x-tex">\\young(ab,c)</annotation>'); |
| 717 | }); |
| 718 | check("Young source survives nested pipe protection without cross-assignment", () => { |
| 719 | const html = renderHtml("$V=\\yng(2,1) | x + \\young(ab,c)$"); |
| 720 | const source = "V=\\yng(2,1) | x + \\young(ab,c)"; |
| 721 | return html.includes("katex") |
| 722 | && !html.includes("katex-error") |
| 723 | && !html.includes("reasonixInternal") |
| 724 | && html.includes(`data-latex-source="${source}"`) |
| 725 | && html.includes(`encoding="application/x-tex">${source}</annotation>`); |
| 726 | }); |
| 727 | check("multiple formulas restore their own source without cross-assignment", () => { |
| 728 | const html = renderHtml("$|x|$ then $x = 50%$"); |
| 729 | const annotations = html.match(/<annotation encoding="application\/x-tex">.*?<\/annotation>/g) ?? []; |
| 730 | return annotations.length === 2 |
| 731 | && annotations[0].includes(">|x|</annotation>") |
| 732 | && annotations[1].includes(">x = 50%</annotation>"); |
| 733 | }); |
| 734 | check("blockquote display math does not swallow following inline math", () => { |
| 735 | const html = renderHtml("> theorem\n> $$E=mc^2$$\n> after $x$"); |
| 736 | return html.includes("katex-display") |
| 737 | && !html.includes("katex-error") |
| 738 | && html.includes(">x</mi>"); |
| 739 | }); |
| 740 | check("multi-line blockquote display math strips quote markers from the formula", () => { |
| 741 | const html = renderHtml("> theorem\n> $$\n> E=mc^2\n> $$\n> after $x$"); |
| 742 | return html.includes("katex-display") |
| 743 | && !html.includes("katex-error") |
| 744 | && !html.includes("> E") |
| 745 | && html.includes(">x</mi>"); |
| 746 | }); |
| 747 | |
| 748 | // ── Summary ─────────────────────────────────────────────────────────────────── |
| 749 | |
| 750 | console.log(`\n${passed} passed, ${failed} failed, ${passed + failed} total`); |
| 751 | if (failed > 0) process.exit(1); |
| 752 |