返回 DeepSeek-Reasonix
mathNormalize.ts
根目录 / desktop / frontend / src / components / mathNormalize.ts
1 // Deterministic pre-pass that repairs LLM-typical math syntax before
2 // remark-math parses Markdown. Semantic inline classification lives in
3 // remarkMathPolicy, where code boundaries and surrounding AST context are
4 // already known.
5 //
6 // 1. Protect Markdown code spans/fences from all math rewrites.
7 // 2. Protect LaTeX line-break spacing (\\[...]) from the LLM-delimiter rewrite.
8 // 3. \(...)/\[...] → $/$$.
9 // 4. Expand \yng/\young to KaTeX-compatible \boxed{array} forms.
10 // Stateful: tracks `$…$` so bare macros in prose get wrapped in
11 // `$…$` and macros already inside math just substitute.
12 // 5. Inline `$$` glued to prose gets a blank line inserted before it
13 // (CommonMark requires that block math be paragraph-separated).
14 // 6. Escape currency dollars (`$5` followed by prose) so greedy single-$
15 // pairing cannot swallow a later math span or pair across amounts
16 // (assistant-ui escapeCurrencyDollars parity).
17 // 7. Protect pipes inside inline math before GFM table tokenisation.
18 // 8. Restore placeholders for remark-math. KaTeX-specific normalisation is
19 // handled by the AST policy after parsing.
20
21 import { expandYoungDiagrams } from "./youngDiagrams";
22
23 // Matches $\cmd{...}...$ where the body may contain $ and one level of nested
24 // braces. Group 1 captures the full \cmd{...} including the outer }. After
25 // the closing }, [^$]*? consumes any trailing content (e.g. " + x^2") up to
26 // the closing $, so patterns like $\text{a} + x^2$ are handled as a whole
27 // rather than split at stray $ signs inside \text{}.
28 const TEXT_MODE_PAIR = /\$\s*(\\[A-Za-z]+\{(?:[^{}]|\{[^{}]*\})*\}[^$]*?)\s*\$/g;
29
30 const DM = "__REASONIX_MATH_DISPLAY__";
31 const IM = "__REASONIX_MATH_INLINE__";
32 const LB = "__REASONIX_LATEX_LINEBREAK__";
33 const ED_BASE = "REASONIXESCAPEDDOLLAR";
34 const INLINE_SOURCE_PREFIX = "\\reasonixInternalSourceV1{";
35 const INLINE_RENDER_SOURCE_PREFIX = "\\reasonixInternalRenderV1";
36 const INLINE_RENDER_SOURCE_RE = /\\reasonixInternalRenderV1\{([^}]*)\}\{([^}]*)\}/g;
37
38 export function normalizeMath(s: string): string {
39 const protectedCode = protectMarkdownCode(s);
40 let r = normalizeMathText(protectedCode.text);
41 for (let i = 0; i < protectedCode.segments.length; i += 1) {
42 r = r.split(`${protectedCode.prefix}${i}__`).join(protectedCode.segments[i]);
43 }
44 return r;
45 }
46
47 function normalizeMathText(s: string): string {
48 // Step 1: protect LaTeX line-break spacing (\\[4pt], \\[2ex], ...) so the
49 // \[ → $$ rewrite below doesn't swallow it.
50 let r = s.replace(/\\\\\[/g, LB);
51
52 // Step 2: convert LLM-native delimiters to standard $/$$ syntax. Arrow
53 // functions are required because "$$" in a JS replace string means a
54 // single literal $.
55 r = r
56 .replace(/\\\[/g, () => "$$")
57 .replace(/\\\]/g, () => "$$")
58 .replace(/\\\(/g, () => "$")
59 .replace(/\\\)/g, () => "$");
60 r = r.replace(new RegExp(LB, "g"), "\\\\[");
61
62 // Step 2.5: expand \yng/\young macros to KaTeX-compatible \boxed{array}
63 // forms after LLM-native delimiters have been converted, so macros inside
64 // \(...\) or \[...\] are correctly recognised as already being in math.
65 r = expandYoungDiagrams(r, ({ source, rendered }) => {
66 return protectInlineMathRenderSource(source, rendered);
67 });
68
69 // Escaped dollars are literal prose dollars, not math delimiters. Hide them
70 // while the structural dollar scans below run, then restore them before
71 // remark-math parses the result.
72 const escapedDollarToken = unusedEscapedDollarToken(r);
73 r = r.split("\\$").join(escapedDollarToken);
74
75 // Step 3+4: normalise display $$ block structure. remark-math requires
76 // opening and closing $$ to sit on their own lines; LLMs often emit
77 // single-line displays, opening fences glued to prose, and adjacent display
78 // blocks separated by prose. A line parser avoids the old cross-block regex
79 // capture that swallowed prose between two display blocks.
80 r = normaliseDisplayBlocks(r);
81
82 // Step 5: preserve the complete source of $\cmd{...}$ pairs containing a
83 // stray inner $ (e.g. $\text{price is $5}$) in a parser-safe marker. The
84 // AST policy restores and normalises it after remark-math has established
85 // the real formula boundary.
86 r = r.replace(TEXT_MODE_PAIR, (match, math: string) => {
87 return math.includes("$") ? `${IM}${protectInlineMathSource(math)}${IM}` : match;
88 });
89
90 // Step 5.5: escape currency dollars so a stray amount cannot pair with a
91 // later math span's opening delimiter. remark-math pairs any two `$` on a
92 // line, so `budget is $100 and the answer is $42$` would otherwise merge
93 // `$100 and the answer is $` into one junk span and destroy the math. A
94 // single `$` immediately followed by a digit is prose currency unless the
95 // span up to the next `$` reads as a math body (assistant-ui
96 // `escapeCurrencyDollars` parity). Display `$$` runs and escaped dollars
97 // (already hidden above) are never touched.
98 r = escapeCurrencyDollars(r, escapedDollarToken);
99
100 // Step 6: GFM identifies table cells before remark plugins can transform the
101 // AST. Normalise only inline spans containing pipes at this syntax boundary
102 // so formulas such as `$|x|$` cannot be split into separate cells. A pipe is
103 // already an explicit math signal in the semantic classifier; all other
104 // inline decisions remain deferred to remarkMathPolicy.
105 r = protectInlineMathPipesForGfm(r);
106
107 // Step 7: restore standard $/$$ delimiters for remark-math to parse.
108 return r
109 .replace(new RegExp(DM, "g"), () => "$$")
110 .replace(new RegExp(IM, "g"), "$")
111 .split(escapedDollarToken).join("\\$");
112 }
113
114 function protectInlineMathPipesForGfm(s: string): string {
115 return s.replace(/\$([^$\n]+)\$/g, (match, math: string) => {
116 let hasUnescapedPipe = false;
117 let backslashes = 0;
118 for (const char of math) {
119 if (char === "|" && backslashes % 2 === 0) {
120 hasUnescapedPipe = true;
121 break;
122 }
123 backslashes = char === "\\" ? backslashes + 1 : 0;
124 }
125 return hasUnescapedPipe ? `$${protectInlineMathSource(math)}$` : match;
126 });
127 }
128
129 // ── currency-dollar escaping (assistant-ui escapeCurrencyDollars parity) ─────
130
131 const MATH_BODY_BLANK_LINE = /\n\s*\n/;
132 const MATH_BODY_LATEX = /\\[A-Za-z]+|[\^_{}]/;
133 const MATH_BODY_ADJACENT_WORDS = /[A-Za-z0-9]\s+[A-Za-z0-9]/;
134 const MATH_BODY_DANGLING_OPERATOR = /[+\-*/=<>~^_]$/;
135
136 /**
137 * Whether the text between two single `$` reads as an inline math expression
138 * rather than the prose separating two currency amounts. A body that ends
139 * the way prose between amounts does (mid-sentence space, dangling operator)
140 * is rejected even when it carries LaTeX-looking syntax, since such prose may
141 * itself contain `_` or `\word`; otherwise LaTeX syntax accepts the span and
142 * two adjacent words reject it.
143 */
144 function isCurrencySafeMathBody(body: string): boolean {
145 if (body.length === 0) return false;
146 if (MATH_BODY_BLANK_LINE.test(body)) return false;
147 if (/\s$/.test(body) && !/^\s/.test(body)) return false;
148 if (MATH_BODY_DANGLING_OPERATOR.test(body)) return false;
149 if (MATH_BODY_LATEX.test(body)) return true;
150 return !MATH_BODY_ADJACENT_WORDS.test(body);
151 }
152
153 function isDigitChar(char: string | undefined): boolean {
154 return char !== undefined && char >= "0" && char <= "9";
155 }
156
157 function dollarRunLength(s: string, index: number): number {
158 let length = 0;
159 while (s[index + length] === "$") length += 1;
160 return length;
161 }
162
163 function nextSingleDollar(s: string, from: number): number {
164 let index = from;
165 while (index < s.length) {
166 if (s[index] === "$") {
167 // A `$$` run is a display delimiter, not an inline closer.
168 return dollarRunLength(s, index) >= 2 ? -1 : index;
169 }
170 index += 1;
171 }
172 return -1;
173 }
174
175 /**
176 * Escapes a `$` that opens a currency amount (`$5`, `$19.99`, `$1,299`) so
177 * remark-math's greedy single-dollar pairing cannot consume it — either as a
178 * junk span with a later math delimiter or as one half of a cross-amount
179 * pair. A `$` followed by a digit is currency unless the span up to the next
180 * `$` is a plausible math body, so `$42$` and `$5x = 10$` survive while
181 * `$5 and $7` and `budget is $100 … $42$` stay literal/functional.
182 * Escaping via `escapedDollarToken` keeps the emitted text free of `$` so
183 * later `$`-scanning steps (pipe protection, remark-math) cannot re-pair it.
184 */
185 function escapeCurrencyDollars(s: string, escapedDollar: string): string {
186 let out = "";
187 let index = 0;
188 while (index < s.length) {
189 const char = s[index];
190 // Escaped characters (LaTeX commands, \\) pass through untouched.
191 if (char === "\\") {
192 out += s.slice(index, index + 2);
193 index += 2;
194 continue;
195 }
196 if (char !== "$") {
197 out += char;
198 index += 1;
199 continue;
200 }
201 const run = dollarRunLength(s, index);
202 if (run >= 2) {
203 out += s.slice(index, index + run);
204 index += run;
205 continue;
206 }
207 const close = nextSingleDollar(s, index + 1);
208 const body = close === -1 ? null : s.slice(index + 1, close);
209 const opensMath = body !== null
210 && !isDigitChar(s[close + 1])
211 && isCurrencySafeMathBody(body);
212 if (opensMath) {
213 out += s.slice(index, close + 1);
214 index = close + 1;
215 continue;
216 }
217 out += isDigitChar(s[index + 1]) ? escapedDollar : "$";
218 index += 1;
219 }
220 return out;
221 }
222
223 function protectInlineMathSource(source: string): string {
224 return `${INLINE_SOURCE_PREFIX}${encodeURIComponent(source)}}`;
225 }
226
227 function protectInlineMathRenderSource(source: string, rendered: string): string {
228 return `${INLINE_RENDER_SOURCE_PREFIX}{${encodeURIComponent(source)}}{${encodeURIComponent(rendered)}}`;
229 }
230
231 export interface ResolvedInlineMathSource {
232 source: string;
233 rendered: string;
234 }
235
236 export function resolveProtectedInlineMathSource(value: string): ResolvedInlineMathSource {
237 let payload = value;
238 if (payload.startsWith(INLINE_SOURCE_PREFIX) && payload.endsWith("}")) {
239 const encoded = payload.slice(INLINE_SOURCE_PREFIX.length, -1);
240 try {
241 payload = decodeURIComponent(encoded);
242 } catch {
243 payload = value;
244 }
245 }
246
247 const replaceRenderSource = (part: "source" | "rendered"): string => {
248 return payload.replace(
249 INLINE_RENDER_SOURCE_RE,
250 (match, encodedSource: string, encodedRendered: string) => {
251 try {
252 return decodeURIComponent(part === "source" ? encodedSource : encodedRendered);
253 } catch {
254 return match;
255 }
256 },
257 );
258 };
259
260 return {
261 source: replaceRenderSource("source"),
262 rendered: replaceRenderSource("rendered"),
263 };
264 }
265
266 export function restoreProtectedInlineMathSource(source: string): string {
267 return resolveProtectedInlineMathSource(source).source;
268 }
269
270 function unusedEscapedDollarToken(s: string): string {
271 let token = ED_BASE;
272 let n = 0;
273 while (s.includes(token)) {
274 n += 1;
275 token = `${ED_BASE}${n}`;
276 }
277 return token;
278 }
279
280 function protectMarkdownCode(s: string): { text: string; prefix: string; segments: string[] } {
281 const prefix = unusedPlaceholderPrefix(s);
282 const segments: string[] = [];
283 let out = "";
284 let i = 0;
285
286 const pushSegment = (segment: string) => {
287 const token = `${prefix}${segments.length}__`;
288 segments.push(segment);
289 out += token;
290 };
291
292 while (i < s.length) {
293 const fenceEnd = fencedCodeEnd(s, i);
294 if (fenceEnd > i) {
295 pushSegment(s.slice(i, fenceEnd));
296 i = fenceEnd;
297 continue;
298 }
299
300 if (s[i] === "`") {
301 const tickEnd = inlineCodeEnd(s, i);
302 if (tickEnd > i) {
303 pushSegment(s.slice(i, tickEnd));
304 i = tickEnd;
305 continue;
306 }
307 }
308
309 out += s[i];
310 i += 1;
311 }
312
313 return { text: out, prefix, segments };
314 }
315
316 function unusedPlaceholderPrefix(s: string): string {
317 let prefix = "__REASONIX_PROTECTED_CODE__";
318 let n = 0;
319 while (s.includes(prefix)) {
320 n += 1;
321 prefix = `__REASONIX_PROTECTED_CODE_${n}__`;
322 }
323 return prefix;
324 }
325
326 function fencedCodeEnd(s: string, start: number): number {
327 // Fence must be at the start of a line (or the document) — CommonMark
328 // requirement. Allowing mid-line fences would swallow prose like
329 // "wrap code in ```blocks``` here" into the code region.
330 if (start !== 0 && s[start - 1] !== "\n") return -1;
331
332 let markerStart = start;
333 let spaces = 0;
334 while (spaces < 4 && s[markerStart] === " ") {
335 markerStart += 1;
336 spaces += 1;
337 }
338
339 const marker = s[markerStart];
340 if (marker !== "`" && marker !== "~") return -1;
341
342 let fenceLen = 0;
343 while (s[markerStart + fenceLen] === marker) fenceLen += 1;
344 if (fenceLen < 3) return -1;
345
346 const openingLineEnd = lineEnd(s, markerStart + fenceLen);
347
348 // Single-line doc: treat the next matching fence as the closing fence.
349 if (openingLineEnd >= s.length) {
350 const fencePattern = marker.repeat(fenceLen);
351 const nextFence = s.indexOf(fencePattern, markerStart + fenceLen);
352 if (nextFence === -1) return s.length;
353 return nextFence + fenceLen;
354 }
355
356 let lineStart = openingLineEnd + 1;
357 while (lineStart < s.length) {
358 const currentLineEnd = lineEnd(s, lineStart);
359 if (isClosingFenceLine(s, lineStart, currentLineEnd, marker, fenceLen)) {
360 return currentLineEnd < s.length ? currentLineEnd + 1 : currentLineEnd;
361 }
362 lineStart = currentLineEnd < s.length ? currentLineEnd + 1 : currentLineEnd;
363 }
364
365 return s.length;
366 }
367
368 function isClosingFenceLine(s: string, start: number, end: number, marker: string, minLen: number): boolean {
369 let i = start;
370 let spaces = 0;
371 while (spaces < 4 && s[i] === " ") {
372 i += 1;
373 spaces += 1;
374 }
375
376 let count = 0;
377 while (s[i + count] === marker) count += 1;
378 if (count < minLen) return false;
379
380 for (let j = i + count; j < end; j += 1) {
381 if (s[j] !== " " && s[j] !== "\t") return false;
382 }
383 return true;
384 }
385
386 function inlineCodeEnd(s: string, start: number): number {
387 let tickLen = 0;
388 while (s[start + tickLen] === "`") tickLen += 1;
389
390 const ticks = "`".repeat(tickLen);
391 const end = s.indexOf(ticks, start + tickLen);
392 return end < 0 ? -1 : end + tickLen;
393 }
394
395 function lineEnd(s: string, start: number): number {
396 const end = s.indexOf("\n", start);
397 return end < 0 ? s.length : end;
398 }
399
400 function normaliseDisplayBlocks(s: string): string {
401 const lines = s.split("\n");
402 const out: string[] = [];
403 let i = 0;
404
405 while (i < lines.length) {
406 const line = lines[i];
407 const $$idx = line.indexOf("$$");
408
409 if ($$idx >= 0 && line.indexOf("$$", $$idx + 2) >= 0
410 && !($$idx > 0 && /\d/.test(line[$$idx - 1]))) {
411 const m = line.match(/^(.*?)\$\$([^\n]*?)\$\$(.*)$/);
412 if (m) {
413 const quote = blockquotePrefix(m[1]);
414 pushDisplayBefore(out, m[1]);
415 out.push(DM);
416 out.push(m[2]);
417 out.push(DM);
418 if (m[3]) out.push(normaliseDisplayBlocks(quote ? quote + m[3].trimStart() : m[3]));
419 i += 1;
420 continue;
421 }
422 }
423
424 if ($$idx >= 0 && line.indexOf("$$", $$idx + 2) < 0
425 && !($$idx > 0 && /\d/.test(line[$$idx - 1]))) {
426 const before = line.slice(0, $$idx);
427 const afterOpen = line.slice($$idx + 2);
428 const quote = blockquotePrefix(before);
429
430 const formulaLines: string[] = [];
431 if (afterOpen) formulaLines.push(afterOpen);
432
433 let j = i + 1;
434 let found = false;
435 while (j < lines.length) {
436 const rawLine = lines[j];
437 const fLine = quote ? stripBlockquotePrefix(rawLine, quote) : rawLine;
438 const closeIdx = fLine.indexOf("$$");
439 if (closeIdx >= 0 && fLine.indexOf("$$", closeIdx + 2) < 0) {
440 const formulaPart = fLine.slice(0, closeIdx);
441 const afterClose = fLine.slice(closeIdx + 2);
442 pushDisplayBefore(out, before);
443 if (formulaPart) formulaLines.push(formulaPart);
444 const formula = formulaLines.join("\n");
445 out.push(DM);
446 out.push(formula);
447 out.push(DM);
448 if (afterClose) out.push(quote ? quote + afterClose.trimStart() : afterClose);
449 i = j + 1;
450 found = true;
451 break;
452 }
453 formulaLines.push(fLine);
454 j += 1;
455 }
456
457 if (found) continue;
458 pushDisplayBefore(out, before);
459 out.push("$$");
460 if (afterOpen) out.push(afterOpen);
461 i += 1;
462 continue;
463 }
464
465 out.push(line);
466 i += 1;
467 }
468
469 return out.join("\n");
470 }
471
472 function pushDisplayBefore(out: string[], before: string): void {
473 if (!before.trim()) return;
474 out.push(before);
475 }
476
477 function blockquotePrefix(before: string): string | null {
478 const m = before.match(/^(\s*>\s*)/);
479 return m ? m[1] : null;
480 }
481
482 function stripBlockquotePrefix(line: string, prefix: string): string {
483 if (line.startsWith(prefix)) return line.slice(prefix.length);
484 const marker = prefix.trimEnd();
485 if (marker && line.startsWith(marker)) {
486 const rest = line.slice(marker.length);
487 return rest.startsWith(" ") ? rest.slice(1) : rest;
488 }
489 return line;
490 }
491
491 lines TYPESCRIPT