mirror of
https://github.com/danny-avila/LibreChat.git
synced 2026-08-27 12:13:30 +00:00
🧮 fix: Currency-Safe Single-Dollar LaTeX via Micromark Tokenizer (#15181)
Replaces the preprocessLaTeX string pass, whose currency allowlist missed suffixes like "$2bn", letting SINGLE_DOLLAR_REGEX rewrite "$2bn to at least $4bn" into $$-math. Single-dollar math is now a micromark text construct (client/src/utils/latex.ts) registered by remarkSingleDollarMath, so each span is decided during parsing with Pandoc-style boundary rules: non-space after the opener, non-space before and no digit after the closer, single-line, no backticks, opaque backslash escapes, balanced braces, and fail-fast on an invalid close so a later price dollar can never extend a span. Rejected spans stay byte-identical text, and code spans, fences, and autolinks are structurally protected by the parser. The LaTeX parsing setting now gates only this plugin; $$, \(...\), and \[...\] continue to parse unconditionally via remark-math (aliased to micromark-extension-llm-math, now mirrored in jest moduleNameMapper so tests exercise the production tokenizer). katex/contrib/mhchem is loaded with the markdown config, so \ce/\pu render properly instead of being regex-mangled. splitMarkdown aligns its math options with the renderer.
This commit is contained in:
parent
6988ff5d7b
commit
c0a55aa0f5
8 changed files with 453 additions and 389 deletions
|
|
@ -25,6 +25,8 @@ module.exports = {
|
|||
// },
|
||||
moduleNameMapper: {
|
||||
'\\.(css)$': 'identity-obj-proxy',
|
||||
/** Mirror the vite resolve.alias so tests parse math with the same tokenizer as production. */
|
||||
'^micromark-extension-math$': 'micromark-extension-llm-math',
|
||||
'\\.(jpg|jpeg|png|gif|eot|otf|webp|svg|ttf|woff|woff2|mp4|webm|wav|mp3|m4a|aac|oga)$':
|
||||
'jest-file-loader',
|
||||
'^test/(.*)$': '<rootDir>/test/$1',
|
||||
|
|
|
|||
|
|
@ -91,6 +91,8 @@
|
|||
"micromark-extension-gfm": "^3.0.0",
|
||||
"micromark-extension-llm-math": "^3.1.0",
|
||||
"micromark-extension-math": "^3.1.0",
|
||||
"micromark-util-character": "^2.1.0",
|
||||
"micromark-util-symbol": "^2.0.0",
|
||||
"monaco-editor": "^0.56.0",
|
||||
"qrcode.react": "^4.2.0",
|
||||
"rc-input-number": "^7.4.2",
|
||||
|
|
@ -159,6 +161,7 @@
|
|||
"jest-environment-jsdom": "^30.2.0",
|
||||
"jest-file-loader": "^1.0.3",
|
||||
"jest-junit": "^17.0.0",
|
||||
"micromark-util-types": "^2.0.0",
|
||||
"postcss": "^8.5.18",
|
||||
"postcss-preset-env": "^11.2.0",
|
||||
"tailwindcss": "^3.4.1",
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
import React, { memo, useMemo, useRef, useEffect } from 'react';
|
||||
import React, { memo, useRef, useEffect } from 'react';
|
||||
import { useRecoilValue } from 'recoil';
|
||||
import { getRemarkPlugins, getRehypePlugins, getMarkdownComponents } from './markdownConfig';
|
||||
import useSmoothStreaming from '~/hooks/Messages/useSmoothStreaming';
|
||||
|
|
@ -6,7 +6,6 @@ import MarkdownErrorBoundary from './MarkdownErrorBoundary';
|
|||
import { FADE_HYDRATION_THRESHOLD } from './animate';
|
||||
import { useMessageContext } from '~/Providers';
|
||||
import MarkdownBlocks from './MarkdownBlocks';
|
||||
import { preprocessLaTeX } from '~/utils';
|
||||
import store from '~/store';
|
||||
|
||||
type TContentProps = {
|
||||
|
|
@ -39,13 +38,6 @@ const Markdown = memo(function Markdown({ content = '', isLatestMessage }: TCont
|
|||
}
|
||||
}, [animate]);
|
||||
|
||||
const currentContent = useMemo(() => {
|
||||
if (isInitializing) {
|
||||
return '';
|
||||
}
|
||||
return LaTeXParsing ? preprocessLaTeX(content) : content;
|
||||
}, [content, LaTeXParsing, isInitializing]);
|
||||
|
||||
if (isInitializing) {
|
||||
return (
|
||||
<div className="absolute">
|
||||
|
|
@ -59,8 +51,8 @@ const Markdown = memo(function Markdown({ content = '', isLatestMessage }: TCont
|
|||
return (
|
||||
<MarkdownErrorBoundary content={content} codeExecution={true}>
|
||||
<MarkdownBlocks
|
||||
content={currentContent}
|
||||
remarkPlugins={getRemarkPlugins()}
|
||||
content={content}
|
||||
remarkPlugins={getRemarkPlugins(LaTeXParsing)}
|
||||
rehypePlugins={getRehypePlugins()}
|
||||
components={getMarkdownComponents()}
|
||||
animate={animate}
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import 'katex/contrib/mhchem';
|
||||
import remarkGfm from 'remark-gfm';
|
||||
import remarkMath from 'remark-math';
|
||||
import supersub from 'remark-supersub';
|
||||
|
|
@ -12,9 +13,9 @@ import {
|
|||
MCPUIResourceCarousel,
|
||||
} from '~/components/MCPUIResource';
|
||||
import { Citation, CompositeCitation, HighlightedText } from '~/components/Web/Citation';
|
||||
import { langSubset, remarkApproxTilde, remarkSingleDollarMath } from '~/utils';
|
||||
import { Artifact, artifactPlugin } from '~/components/Artifacts/Artifact';
|
||||
import { code, a, p, img, table } from './MarkdownComponents';
|
||||
import { langSubset, remarkApproxTilde } from '~/utils';
|
||||
import { unicodeCitation } from '~/components/Web';
|
||||
|
||||
/**
|
||||
|
|
@ -30,13 +31,19 @@ import { unicodeCitation } from '~/components/Web';
|
|||
* read to call time (when components render or memoize) sidesteps the
|
||||
* temporal dead zone.
|
||||
*/
|
||||
export const getRemarkPlugins = (): PluggableList => [
|
||||
/**
|
||||
* `latexParsing` (the user setting) gates only the ambiguous single-dollar syntax; the
|
||||
* unambiguous `$$`, `\(...\)`, and `\[...\]` delimiters always parse via `remark-math`
|
||||
* (aliased to `micromark-extension-llm-math` in vite and jest config).
|
||||
*/
|
||||
export const getRemarkPlugins = (latexParsing = true): PluggableList => [
|
||||
remarkApproxTilde,
|
||||
supersub,
|
||||
remarkGfm,
|
||||
remarkDirective,
|
||||
artifactPlugin,
|
||||
[remarkMath, { singleDollarTextMath: false }],
|
||||
...(latexParsing ? [remarkSingleDollarMath] : []),
|
||||
unicodeCitation,
|
||||
mcpUIResourcePlugin,
|
||||
];
|
||||
|
|
|
|||
|
|
@ -92,11 +92,12 @@ const countWithin = (
|
|||
* render pipeline relies on (GFM tables, container directives like
|
||||
* `:::artifact:::`, and `$$` math), so top-level block boundaries match what
|
||||
* react-markdown produces. Inline-only transforms (citations, MCP-UI markers,
|
||||
* supersub) never cross a top-level block, so they are intentionally omitted.
|
||||
* supersub, single-dollar math) never cross a top-level block, so they are
|
||||
* intentionally omitted.
|
||||
*/
|
||||
const parseToMdast = (content: string): MdastNode =>
|
||||
fromMarkdown(content, {
|
||||
extensions: [gfm(), directive(), math()],
|
||||
extensions: [gfm(), directive(), math({ singleDollarTextMath: false })],
|
||||
mdastExtensions: [gfmFromMarkdown(), directiveFromMarkdown(), mathFromMarkdown()],
|
||||
}) as MdastNode;
|
||||
|
||||
|
|
|
|||
|
|
@ -1,254 +1,314 @@
|
|||
import { preprocessLaTeX } from './latex';
|
||||
import { createElement } from 'react';
|
||||
import ReactMarkdown from 'react-markdown';
|
||||
import { render } from '@testing-library/react';
|
||||
import { math } from 'micromark-extension-math';
|
||||
import { mathFromMarkdown } from 'mdast-util-math';
|
||||
import { fromMarkdown } from 'mdast-util-from-markdown';
|
||||
import type { Options as ReactMarkdownOptions } from 'react-markdown';
|
||||
import {
|
||||
getRemarkPlugins,
|
||||
getRehypePlugins,
|
||||
} from '~/components/Chat/Messages/Content/markdownConfig';
|
||||
import { singleDollarMath } from './latex';
|
||||
|
||||
describe('preprocessLaTeX', () => {
|
||||
test('returns the same string if no LaTeX patterns are found', () => {
|
||||
const content = 'This is a test string without LaTeX or dollar signs';
|
||||
expect(preprocessLaTeX(content)).toBe(content);
|
||||
type SpecNode = {
|
||||
type: string;
|
||||
value?: string;
|
||||
children?: SpecNode[];
|
||||
};
|
||||
|
||||
/**
|
||||
* Mirrors the production parser: `micromark-extension-math` resolves to
|
||||
* `micromark-extension-llm-math` (vite alias in the app, moduleNameMapper here), with
|
||||
* single-dollar spans handled exclusively by the `singleDollarMath` construct.
|
||||
*/
|
||||
const parse = (content: string): SpecNode =>
|
||||
fromMarkdown(content, {
|
||||
extensions: [math({ singleDollarTextMath: false }), singleDollarMath],
|
||||
mdastExtensions: [mathFromMarkdown()],
|
||||
}) as SpecNode;
|
||||
|
||||
const collect = (node: SpecNode, type: string, values: string[] = []): string[] => {
|
||||
if (node.type === type && node.value !== undefined) {
|
||||
values.push(node.value);
|
||||
}
|
||||
for (const child of node.children ?? []) {
|
||||
collect(child, type, values);
|
||||
}
|
||||
return values;
|
||||
};
|
||||
|
||||
const hasType = (node: SpecNode, type: string): boolean => {
|
||||
if (node.type === type) {
|
||||
return true;
|
||||
}
|
||||
return (node.children ?? []).some((child) => hasType(child, type));
|
||||
};
|
||||
|
||||
const inlineMath = (content: string): string[] => collect(parse(content), 'inlineMath');
|
||||
const flowMath = (content: string): string[] => collect(parse(content), 'math');
|
||||
const textOf = (content: string): string => collect(parse(content), 'text').join('');
|
||||
|
||||
describe('singleDollarMath', () => {
|
||||
describe('currency stays literal', () => {
|
||||
test('Treasury buyback report (production bug)', () => {
|
||||
const content =
|
||||
'The U.S. Treasury said it would at least double long-dated bond buybacks, from $2bn to at least $4bn per operation starting Sept 9.';
|
||||
expect(inlineMath(content)).toEqual([]);
|
||||
expect(textOf(content)).toContain('from $2bn to at least $4bn per operation');
|
||||
});
|
||||
|
||||
test('plain amounts', () => {
|
||||
expect(inlineMath('Price is $50 and $100')).toEqual([]);
|
||||
expect(inlineMath('$50 is $20 + $30')).toEqual([]);
|
||||
expect(inlineMath('The price is $1,000,000 for this item.')).toEqual([]);
|
||||
expect(inlineMath('Total: $29.50 plus tax')).toEqual([]);
|
||||
});
|
||||
|
||||
test('abbreviated amounts', () => {
|
||||
expect(inlineMath('Revenue: $5M to $10M, funding: $1.5B, price: $5K')).toEqual([]);
|
||||
expect(inlineMath('$250k is 25% of $1M')).toEqual([]);
|
||||
expect(inlineMath('More than $1bn in leveraged shorts, over $3bn total')).toEqual([]);
|
||||
});
|
||||
|
||||
test('long decimals and large numbers', () => {
|
||||
expect(inlineMath('You can win $1000000 or even $9999999.99!')).toEqual([]);
|
||||
expect(inlineMath('Bitcoin: $0.00001234, Gas: $3.999, Rate: $1.234567890')).toEqual([]);
|
||||
expect(
|
||||
inlineMath('The total is $1157.90 (existing) + $500 (new investment) = $1657.90.'),
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
test('ranges with a punctuation dash reject on the trailing digit', () => {
|
||||
expect(inlineMath('a $100-$200 range')).toEqual([]);
|
||||
expect(inlineMath('a $100–$200 range')).toEqual([]);
|
||||
expect(inlineMath('in the $10k-$20k band')).toEqual([]);
|
||||
});
|
||||
|
||||
test('sums across a whole line', () => {
|
||||
expect(inlineMath('- **Total Savings**: $500 + $200 + $150 = $850')).toEqual([]);
|
||||
});
|
||||
|
||||
test('suffixed European style amounts', () => {
|
||||
expect(inlineMath('Cela coûte 100$ et 200$ en Europe')).toEqual([]);
|
||||
});
|
||||
|
||||
test('lone and trailing dollar signs', () => {
|
||||
expect(inlineMath('A single $ sign should not be converted')).toEqual([]);
|
||||
expect(inlineMath('The price hit $79,455 on')).toEqual([]);
|
||||
});
|
||||
|
||||
test('amounts on separate lines of one paragraph', () => {
|
||||
expect(inlineMath('Currency $100 and\nthen $200 later')).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
test('returns the same string if no dollar signs are present', () => {
|
||||
const content = 'This has LaTeX \\(x^2\\) and \\[y^2\\] but no dollars';
|
||||
expect(preprocessLaTeX(content)).toBe(content);
|
||||
describe('single-dollar math parses', () => {
|
||||
test('basic expressions', () => {
|
||||
expect(inlineMath('Inline math: $x^2 + y^2 = z^2$')).toEqual(['x^2 + y^2 = z^2']);
|
||||
expect(inlineMath('Equation: $f(x) = 2x + 3$ where x is a variable.')).toEqual([
|
||||
'f(x) = 2x + 3',
|
||||
]);
|
||||
expect(inlineMath('First $a + b = c$ and second $x^2 + y^2 = z^2$')).toEqual([
|
||||
'a + b = c',
|
||||
'x^2 + y^2 = z^2',
|
||||
]);
|
||||
});
|
||||
|
||||
test('digit-led expressions are still math', () => {
|
||||
expect(
|
||||
inlineMath('- **Goldbach Conjecture**: $2n = p + q$ (every even integer > 2)'),
|
||||
).toEqual(['2n = p + q']);
|
||||
expect(inlineMath('the answer is $3$.')).toEqual(['3']);
|
||||
expect(inlineMath('an eigenvalue of $-1$ is expected')).toEqual(['-1']);
|
||||
});
|
||||
|
||||
test('letters may follow the closer (ordinals)', () => {
|
||||
expect(inlineMath('the $n$th term')).toEqual(['n']);
|
||||
});
|
||||
|
||||
test('trailing punctuation after the closer', () => {
|
||||
expect(inlineMath('The set is defined as $\\{x | x > 0\\}$.')).toEqual(['\\{x | x > 0\\}']);
|
||||
});
|
||||
|
||||
test('physics expressions', () => {
|
||||
const content = [
|
||||
'- **Schrödinger Equation**: $i\\hbar\\frac{\\partial}{\\partial t}|\\psi\\rangle = \\hat{H}|\\psi\\rangle$',
|
||||
'- **Einstein Field Equations**: $G_{\\mu\\nu} = \\frac{8\\pi G}{c^4} T_{\\mu\\nu}$',
|
||||
].join('\n');
|
||||
expect(inlineMath(content)).toEqual([
|
||||
'i\\hbar\\frac{\\partial}{\\partial t}|\\psi\\rangle = \\hat{H}|\\psi\\rangle',
|
||||
'G_{\\mu\\nu} = \\frac{8\\pi G}{c^4} T_{\\mu\\nu}',
|
||||
]);
|
||||
});
|
||||
|
||||
test('nested braces and subscripted products', () => {
|
||||
expect(
|
||||
inlineMath('Totient: $\\phi(n) = n \\prod_{p|n} \\left(1 - \\frac{1}{p}\\right)$'),
|
||||
).toEqual(['\\phi(n) = n \\prod_{p|n} \\left(1 - \\frac{1}{p}\\right)']);
|
||||
});
|
||||
|
||||
test('escaped dollars stay inside the span', () => {
|
||||
expect(inlineMath('Calculate $\\text{Total} = \\$500 + \\$200$')).toEqual([
|
||||
'\\text{Total} = \\$500 + \\$200',
|
||||
]);
|
||||
expect(inlineMath('The formula $f(x) = \\$2x$ represents cost')).toEqual(['f(x) = \\$2x']);
|
||||
});
|
||||
|
||||
test('math and prices coexist', () => {
|
||||
expect(inlineMath('Formula $x^2$ costs $25')).toEqual(['x^2']);
|
||||
expect(inlineMath('LaTeX $x^2$ and price $50')).toEqual(['x^2']);
|
||||
expect(inlineMath('Price $100 then equation $x + y = z$ then another price $50')).toEqual([
|
||||
'x + y = z',
|
||||
]);
|
||||
});
|
||||
|
||||
test('markdown characters inside math never form emphasis', () => {
|
||||
const content = 'terms $a_1 + b_2$ and $c_{i}^{*}$ here';
|
||||
expect(inlineMath(content)).toEqual(['a_1 + b_2', 'c_{i}^{*}']);
|
||||
expect(hasType(parse(content), 'emphasis')).toBe(false);
|
||||
});
|
||||
|
||||
test('mhchem passes through unmangled', () => {
|
||||
expect(inlineMath('$\\ce{H2O}$ and $\\pu{123 J}$')).toEqual(['\\ce{H2O}', '\\pu{123 J}']);
|
||||
});
|
||||
});
|
||||
|
||||
test('preserves valid inline LaTeX delimiters \\(...\\)', () => {
|
||||
const content = 'This is inline LaTeX: \\(x^2 + y^2 = z^2\\)';
|
||||
expect(preprocessLaTeX(content)).toBe(content);
|
||||
describe('structural protection', () => {
|
||||
test('inline code is untouchable', () => {
|
||||
const content = 'Outside $x^2$ and inside code: `$100`';
|
||||
expect(inlineMath(content)).toEqual(['x^2']);
|
||||
expect(collect(parse(content), 'inlineCode')).toEqual(['$100']);
|
||||
});
|
||||
|
||||
test('a span never swallows an inline code marker', () => {
|
||||
const content = 'The error "invalid $lookup namespace" occurs when using `$lookup` operator';
|
||||
expect(inlineMath(content)).toEqual([]);
|
||||
expect(collect(parse(content), 'inlineCode')).toEqual(['$lookup']);
|
||||
});
|
||||
|
||||
test('math and inline code coexist', () => {
|
||||
const content = 'Use $x + y$ in math but `$lookup` in code';
|
||||
expect(inlineMath(content)).toEqual(['x + y']);
|
||||
expect(collect(parse(content), 'inlineCode')).toEqual(['$lookup']);
|
||||
});
|
||||
|
||||
test('fenced code is untouchable', () => {
|
||||
const content = '```\n$100\n$variable\n```\n\nOutside $x^2$';
|
||||
expect(inlineMath(content)).toEqual(['x^2']);
|
||||
expect(collect(parse(content), 'code')).toEqual(['$100\n$variable']);
|
||||
});
|
||||
});
|
||||
|
||||
test('preserves valid block LaTeX delimiters \\[...\\]', () => {
|
||||
const content = 'This is block LaTeX: \\[E = mc^2\\]';
|
||||
expect(preprocessLaTeX(content)).toBe(content);
|
||||
describe('escapes and line boundaries', () => {
|
||||
test('escaped dollars never open a span', () => {
|
||||
expect(inlineMath('Already escaped \\$50 and \\$100')).toEqual([]);
|
||||
expect(inlineMath('Escaped \\$x^2\\$ should not change')).toEqual([]);
|
||||
});
|
||||
|
||||
test('single-dollar spans never cross lines', () => {
|
||||
expect(inlineMath('This has $x\ny$ which spans lines')).toEqual([]);
|
||||
});
|
||||
|
||||
test('a dangling escape abandons the span', () => {
|
||||
expect(inlineMath('dangling $a\\')).toEqual([]);
|
||||
expect(inlineMath('dangling $a\\\nnext line$')).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
test('preserves valid double dollar delimiters', () => {
|
||||
const content = 'This is valid: $$x^2 + y^2 = z^2$$';
|
||||
expect(preprocessLaTeX(content)).toBe(content);
|
||||
describe('unambiguous delimiters are unaffected', () => {
|
||||
test('double dollars, inline and flow', () => {
|
||||
expect(inlineMath('This is valid: $$x^2 + y^2 = z^2$$')).toEqual(['x^2 + y^2 = z^2']);
|
||||
expect(flowMath('$$\nE = mc^2\n$$')).toEqual(['E = mc^2']);
|
||||
});
|
||||
|
||||
test('TeX brackets from the llm-math fork', () => {
|
||||
expect(inlineMath('This is inline LaTeX: \\(x^2 + y^2 = z^2\\)')).toEqual([
|
||||
'x^2 + y^2 = z^2',
|
||||
]);
|
||||
const display = parse('\\[\nE = mc^2\n\\]');
|
||||
expect([...collect(display, 'math'), ...collect(display, 'inlineMath')]).toEqual([
|
||||
'E = mc^2',
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
test('converts single dollar delimiters to double dollars', () => {
|
||||
const content = 'Inline math: $x^2 + y^2 = z^2$';
|
||||
const expected = 'Inline math: $$x^2 + y^2 = z^2$$';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
describe('documented ambiguity limits', () => {
|
||||
test('a trailing dollar-wrapped number still parses (Pandoc parity)', () => {
|
||||
expect(inlineMath('Simple Interest: $A = P + Prt = $1,000 and = $1,100$')).toEqual(['1,100']);
|
||||
});
|
||||
|
||||
test('converts multiple single dollar expressions', () => {
|
||||
const content = 'First $a + b = c$ and second $x^2 + y^2 = z^2$';
|
||||
const expected = 'First $$a + b = c$$ and second $$x^2 + y^2 = z^2$$';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('escapes currency dollar signs', () => {
|
||||
const content = 'Price is $50 and $100';
|
||||
const expected = 'Price is \\$50 and \\$100';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('escapes currency with spaces', () => {
|
||||
const content = '$50 is $20 + $30';
|
||||
const expected = '\\$50 is \\$20 + \\$30';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('escapes currency with commas', () => {
|
||||
const content = 'The price is $1,000,000 for this item.';
|
||||
const expected = 'The price is \\$1,000,000 for this item.';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('escapes currency with decimals', () => {
|
||||
const content = 'Total: $29.50 plus tax';
|
||||
const expected = 'Total: \\$29.50 plus tax';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('converts LaTeX expressions while escaping currency', () => {
|
||||
const content = 'LaTeX $x^2$ and price $50';
|
||||
const expected = 'LaTeX $$x^2$$ and price \\$50';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles Goldbach Conjecture example', () => {
|
||||
const content = '- **Goldbach Conjecture**: $2n = p + q$ (every even integer > 2)';
|
||||
const expected = '- **Goldbach Conjecture**: $$2n = p + q$$ (every even integer > 2)';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('does not escape already escaped dollar signs', () => {
|
||||
const content = 'Already escaped \\$50 and \\$100';
|
||||
expect(preprocessLaTeX(content)).toBe(content);
|
||||
});
|
||||
|
||||
test('does not convert already escaped single dollars', () => {
|
||||
const content = 'Escaped \\$x^2\\$ should not change';
|
||||
expect(preprocessLaTeX(content)).toBe(content);
|
||||
});
|
||||
|
||||
test('escapes mhchem commands', () => {
|
||||
const content = '$\\ce{H2O}$ and $\\pu{123 J}$';
|
||||
const expected = '$$\\\\ce{H2O}$$ and $$\\\\pu{123 J}$$';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles empty string', () => {
|
||||
expect(preprocessLaTeX('')).toBe('');
|
||||
});
|
||||
|
||||
test('handles complex mixed content', () => {
|
||||
const content = `Valid double $$y^2$$
|
||||
Currency $100 and $200
|
||||
Single dollar math $x^2 + y^2$
|
||||
Chemical $\\ce{H2O}$
|
||||
Valid brackets \\[z^2\\]`;
|
||||
const expected = `Valid double $$y^2$$
|
||||
Currency \\$100 and \\$200
|
||||
Single dollar math $$x^2 + y^2$$
|
||||
Chemical $$\\\\ce{H2O}$$
|
||||
Valid brackets \\[z^2\\]`;
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles multiple equations with currency', () => {
|
||||
const content = `- **Euler's Totient Function**: $\\phi(n) = n \\prod_{p|n} \\left(1 - \\frac{1}{p}\\right)$
|
||||
- **Total Savings**: $500 + $200 + $150 = $850`;
|
||||
const expected = `- **Euler's Totient Function**: $$\\phi(n) = n \\prod_{p|n} \\left(1 - \\frac{1}{p}\\right)$$
|
||||
- **Total Savings**: \\$500 + \\$200 + \\$150 = \\$850`;
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles inline code blocks', () => {
|
||||
const content = 'Outside $x^2$ and inside code: `$100`';
|
||||
const expected = 'Outside $$x^2$$ and inside code: `$100`';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles multiline code blocks', () => {
|
||||
const content = '```\n$100\n$variable\n```\nOutside $x^2$';
|
||||
const expected = '```\n$100\n$variable\n```\nOutside $$x^2$$';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('preserves LaTeX expressions with special characters', () => {
|
||||
const content = 'The set is defined as $\\{x | x > 0\\}$.';
|
||||
const expected = 'The set is defined as $$\\{x | x > 0\\}$$.';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles complex physics equations', () => {
|
||||
const content = `- **Schrödinger Equation**: $i\\hbar\\frac{\\partial}{\\partial t}|\\psi\\rangle = \\hat{H}|\\psi\\rangle$
|
||||
- **Einstein Field Equations**: $G_{\\mu\\nu} = \\frac{8\\pi G}{c^4} T_{\\mu\\nu}$`;
|
||||
const expected = `- **Schrödinger Equation**: $$i\\hbar\\frac{\\partial}{\\partial t}|\\psi\\rangle = \\hat{H}|\\psi\\rangle$$
|
||||
- **Einstein Field Equations**: $$G_{\\mu\\nu} = \\frac{8\\pi G}{c^4} T_{\\mu\\nu}$$`;
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles financial calculations with currency', () => {
|
||||
const content = `- **Simple Interest**: $A = P + Prt = $1,000 + ($1,000)(0.05)(2) = $1,100$
|
||||
- **ROI**: $\\text{ROI} = \\frac{$1,200 - $1,000}{$1,000} \\times 100\\% = 20\\%$`;
|
||||
const expected = `- **Simple Interest**: $$A = P + Prt = \\$1,000 + (\\$1,000)(0.05)(2) = \\$1,100$$
|
||||
- **ROI**: $$\\text{ROI} = \\frac{\\$1,200 - \\$1,000}{\\$1,000} \\times 100\\% = 20\\%$$`;
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('does not convert partial or malformed expressions', () => {
|
||||
const content = 'A single $ sign should not be converted';
|
||||
const expected = 'A single $ sign should not be converted';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles nested parentheses in LaTeX', () => {
|
||||
const content =
|
||||
'Matrix determinant: $\\det(A) = \\sum_{\\sigma \\in S_n} \\text{sgn}(\\sigma) \\prod_{i=1}^n a_{i,\\sigma(i)}$';
|
||||
const expected =
|
||||
'Matrix determinant: $$\\det(A) = \\sum_{\\sigma \\in S_n} \\text{sgn}(\\sigma) \\prod_{i=1}^n a_{i,\\sigma(i)}$$';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('preserves spacing in equations', () => {
|
||||
const content = 'Equation: $f(x) = 2x + 3$ where x is a variable.';
|
||||
const expected = 'Equation: $$f(x) = 2x + 3$$ where x is a variable.';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles LaTeX with newlines inside should not be converted', () => {
|
||||
const content = `This has $x
|
||||
y$ which spans lines`;
|
||||
const expected = `This has $x
|
||||
y$ which spans lines`;
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles multiple dollar signs in text', () => {
|
||||
const content = 'Price $100 then equation $x + y = z$ then another price $50';
|
||||
const expected = 'Price \\$100 then equation $$x + y = z$$ then another price \\$50';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles complex LaTeX with currency in same expression', () => {
|
||||
const content = 'Calculate $\\text{Total} = \\$500 + \\$200$';
|
||||
const expected = 'Calculate $$\\text{Total} = \\$500 + \\$200$$';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('preserves already escaped dollars in LaTeX', () => {
|
||||
const content = 'The formula $f(x) = \\$2x$ represents cost';
|
||||
const expected = 'The formula $$f(x) = \\$2x$$ represents cost';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles adjacent LaTeX and currency', () => {
|
||||
const content = 'Formula $x^2$ costs $25';
|
||||
const expected = 'Formula $$x^2$$ costs \\$25';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles LaTeX with special characters and currency', () => {
|
||||
const content = 'Set $\\{x | x > \\$0\\}$ for positive prices';
|
||||
const expected = 'Set $$\\{x | x > \\$0\\}$$ for positive prices';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('does not convert when closing dollar is preceded by backtick', () => {
|
||||
const content = 'The error "invalid $lookup namespace" occurs when using `$lookup` operator';
|
||||
const expected = 'The error "invalid $lookup namespace" occurs when using `$lookup` operator';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles mixed backtick and non-backtick cases', () => {
|
||||
const content = 'Use $x + y$ in math but `$lookup` in code';
|
||||
const expected = 'Use $$x + y$$ in math but `$lookup` in code';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('escapes currency amounts without commas', () => {
|
||||
const content =
|
||||
'The total amount invested is $1157.90 (existing amount) + $500 (new investment) = $1657.90.';
|
||||
const expected =
|
||||
'The total amount invested is \\$1157.90 (existing amount) + \\$500 (new investment) = \\$1657.90.';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles large currency amounts', () => {
|
||||
const content = 'You can win $1000000 or even $9999999.99!';
|
||||
const expected = 'You can win \\$1000000 or even \\$9999999.99!';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('escapes currency with many decimal places', () => {
|
||||
const content = 'Bitcoin: $0.00001234, Gas: $3.999, Rate: $1.234567890';
|
||||
const expected = 'Bitcoin: \\$0.00001234, Gas: \\$3.999, Rate: \\$1.234567890';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('escapes abbreviated currency notation', () => {
|
||||
const content = '$250k is 25% of $1M';
|
||||
const expected = '\\$250k is 25% of \\$1M';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
});
|
||||
|
||||
test('handles various abbreviated currency formats', () => {
|
||||
const content = 'Revenue: $5M to $10M, funding: $1.5B, price: $5K';
|
||||
const expected = 'Revenue: \\$5M to \\$10M, funding: \\$1.5B, price: \\$5K';
|
||||
expect(preprocessLaTeX(content)).toBe(expected);
|
||||
test('unbalanced braces abandon the span', () => {
|
||||
expect(inlineMath('weird $a}b$ y')).toEqual([]);
|
||||
expect(inlineMath('open $a{b$ y')).toEqual([]);
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('getRemarkPlugins LaTeX wiring', () => {
|
||||
/** The config's unified@10 `PluggableList` and react-markdown's unified@11 plugin types are structurally compatible but nominally distinct, as at the production call sites. */
|
||||
const renderMarkdown = (content: string, latexParsing: boolean) => {
|
||||
const remarkPlugins = getRemarkPlugins(latexParsing) as ReactMarkdownOptions['remarkPlugins'];
|
||||
return render(createElement(ReactMarkdown, { remarkPlugins }, content));
|
||||
};
|
||||
|
||||
test('currency renders literally through the full plugin chain', () => {
|
||||
const { container } = renderMarkdown('from $2bn to at least $4bn per operation', true);
|
||||
expect(container.querySelector('.math-inline')).toBeNull();
|
||||
expect(container.textContent).toContain('from $2bn to at least $4bn per operation');
|
||||
});
|
||||
|
||||
test('single-dollar math renders when enabled', () => {
|
||||
const { container } = renderMarkdown('Equation $E=mc^2$ here', true);
|
||||
const node = container.querySelector('.math-inline');
|
||||
expect(node?.textContent).toBe('E=mc^2');
|
||||
});
|
||||
|
||||
test('the toggle gates only single-dollar syntax', () => {
|
||||
const single = renderMarkdown('Equation $E=mc^2$ here', false);
|
||||
expect(single.container.querySelector('.math-inline')).toBeNull();
|
||||
expect(single.container.textContent).toContain('$E=mc^2$');
|
||||
|
||||
const double = renderMarkdown('Equation $$E=mc^2$$ here', false);
|
||||
expect(double.container.querySelector('.math-inline')).not.toBeNull();
|
||||
|
||||
const brackets = renderMarkdown('Equation \\(E=mc^2\\) here', false);
|
||||
expect(brackets.container.querySelector('.math-inline')).not.toBeNull();
|
||||
});
|
||||
|
||||
test('currency alongside citation anchors stays literal', () => {
|
||||
const content =
|
||||
'The U.S. Treasury said it would at least double long-dated bond buybacks, from $2bn to at least $4bn per operation starting Sept 9. turn0search4 That pushed long-end yields down.';
|
||||
const { container } = renderMarkdown(content, true);
|
||||
expect(container.querySelector('.math-inline')).toBeNull();
|
||||
expect(container.textContent).toContain('from $2bn to at least $4bn per operation');
|
||||
});
|
||||
|
||||
test('KaTeX renders the parsed spans without errors', () => {
|
||||
const rehypePlugins = getRehypePlugins() as ReactMarkdownOptions['rehypePlugins'];
|
||||
const remarkPlugins = getRemarkPlugins(true) as ReactMarkdownOptions['remarkPlugins'];
|
||||
const { container } = render(
|
||||
createElement(
|
||||
ReactMarkdown,
|
||||
{ remarkPlugins, rehypePlugins },
|
||||
'Water is $\\ce{H2O}$ where $E=mc^2$ costs $2bn to at least $4bn.',
|
||||
),
|
||||
);
|
||||
expect(container.querySelectorAll('.katex')).toHaveLength(2);
|
||||
expect(container.querySelector('.katex-error')).toBeNull();
|
||||
expect(container.textContent).toContain('costs $2bn to at least $4bn.');
|
||||
});
|
||||
|
||||
test('currency and approx-tildes inside GFM table cells stay literal', () => {
|
||||
const content = [
|
||||
'| Date | Level | What happened |',
|
||||
'|---|---|---|',
|
||||
'| Aug 19 | ~$64,500 open | Treasury buyback news hits after hours turn0search4 |',
|
||||
'| Aug 21 | $77,300, peak $79,455 | White House Clarity Act push |',
|
||||
].join('\n');
|
||||
const { container } = renderMarkdown(content, true);
|
||||
expect(container.querySelector('.math-inline')).toBeNull();
|
||||
expect(container.textContent).toContain('$64,500 open');
|
||||
expect(container.textContent).toContain('$77,300, peak $79,455');
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1,152 +1,148 @@
|
|||
// Pre-compile all regular expressions for better performance
|
||||
const MHCHEM_CE_REGEX = /\$\\ce\{/g;
|
||||
const MHCHEM_PU_REGEX = /\$\\pu\{/g;
|
||||
const MHCHEM_CE_ESCAPED_REGEX = /\$\\\\ce\{[^}]*\}\$/g;
|
||||
const MHCHEM_PU_ESCAPED_REGEX = /\$\\\\pu\{[^}]*\}\$/g;
|
||||
const CURRENCY_REGEX =
|
||||
/(?<![\\$])\$(?!\$)(?=\d+(?:,\d{3})*(?:\.\d+)?(?:[KMBkmb])?(?:\s|$|[^a-zA-Z\d]))/g;
|
||||
const SINGLE_DOLLAR_REGEX = /(?<!\\)\$(?!\$)((?:[^$\n]|\\[$])+?)(?<!\\)(?<!`)\$(?!\$)/g;
|
||||
import { codes, types } from 'micromark-util-symbol';
|
||||
import { asciiDigit, markdownLineEnding, markdownSpace } from 'micromark-util-character';
|
||||
import type {
|
||||
Code,
|
||||
Construct,
|
||||
Effects,
|
||||
Extension,
|
||||
State,
|
||||
TokenizeContext,
|
||||
} from 'micromark-util-types';
|
||||
import type { Plugin } from 'unified';
|
||||
import type { Root } from 'mdast';
|
||||
|
||||
interface ParserData {
|
||||
micromarkExtensions?: Extension[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Escapes mhchem package notation in LaTeX by converting single dollar delimiters to double dollars
|
||||
* and escaping backslashes in mhchem commands.
|
||||
* Single-dollar inline math is re-enabled here as a micromark construct instead of the
|
||||
* string preprocessing it replaces, because `$...$` is ambiguous with prices ("from $2bn
|
||||
* to at least $4bn") and only the tokenizer can decide a span without rewriting the
|
||||
* message: code spans, fences, and autolinks are structurally excluded, and a rejected
|
||||
* span stays byte-identical text. `remark-math` keeps running with
|
||||
* `singleDollarTextMath: false`; this construct is the sole single-dollar path.
|
||||
*
|
||||
* @param text - The input text containing potential mhchem notation
|
||||
* @returns The processed text with properly escaped mhchem notation
|
||||
* A `$...$` span becomes math only when all of Pandoc's boundary rules hold:
|
||||
* - the opening `$` is immediately followed by a non-space character;
|
||||
* - the closing `$` is immediately preceded by a non-space character and not immediately
|
||||
* followed by a digit (rejects ranges like "$100-$200");
|
||||
* - the span stays on one line, contains no backtick, treats `\`-pairs as opaque
|
||||
* (`\$` stays inside the span), and closes with balanced braces.
|
||||
*
|
||||
* A failed close abandons the whole attempt (`nok`) instead of scanning further, so the
|
||||
* next `$` in "Price is $50 and $100" can never silently extend a span; micromark then
|
||||
* retries the construct at that `$` on its own merits.
|
||||
*/
|
||||
function escapeMhchem(text: string): string {
|
||||
// First escape the backslashes in mhchem commands
|
||||
let result = text.replace(MHCHEM_CE_REGEX, '$\\\\ce{');
|
||||
result = result.replace(MHCHEM_PU_REGEX, '$\\\\pu{');
|
||||
function tokenizeMathSpan(this: TokenizeContext, effects: Effects, ok: State, nok: State): State {
|
||||
let previousCode: Code = null;
|
||||
let braceDepth = 0;
|
||||
|
||||
// Then convert single dollar mhchem to double dollar
|
||||
result = result.replace(MHCHEM_CE_ESCAPED_REGEX, (match) => `$${match}$`);
|
||||
result = result.replace(MHCHEM_PU_ESCAPED_REGEX, (match) => `$${match}$`);
|
||||
function start(code: Code): State | undefined {
|
||||
effects.enter('mathText');
|
||||
effects.enter('mathTextSequence');
|
||||
effects.consume(code);
|
||||
effects.exit('mathTextSequence');
|
||||
return open;
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Efficiently finds all code block regions in the content
|
||||
* @param content The content to analyze
|
||||
* @returns Array of code block regions [start, end]
|
||||
*/
|
||||
function findCodeBlockRegions(content: string): Array<[number, number]> {
|
||||
const regions: Array<[number, number]> = [];
|
||||
let inlineStart = -1;
|
||||
let multilineStart = -1;
|
||||
|
||||
for (let i = 0; i < content.length; i++) {
|
||||
const char = content[i];
|
||||
|
||||
// Check for multiline code blocks
|
||||
function open(code: Code): State | undefined {
|
||||
if (
|
||||
char === '`' &&
|
||||
i + 2 < content.length &&
|
||||
content[i + 1] === '`' &&
|
||||
content[i + 2] === '`'
|
||||
code === codes.eof ||
|
||||
code === codes.dollarSign ||
|
||||
code === codes.graveAccent ||
|
||||
markdownSpace(code) ||
|
||||
markdownLineEnding(code)
|
||||
) {
|
||||
if (multilineStart === -1) {
|
||||
multilineStart = i;
|
||||
i += 2; // Skip the next two backticks
|
||||
} else {
|
||||
regions.push([multilineStart, i + 2]);
|
||||
multilineStart = -1;
|
||||
i += 2;
|
||||
}
|
||||
}
|
||||
// Check for inline code blocks (only if not in multiline)
|
||||
else if (char === '`' && multilineStart === -1) {
|
||||
if (inlineStart === -1) {
|
||||
inlineStart = i;
|
||||
} else {
|
||||
regions.push([inlineStart, i]);
|
||||
inlineStart = -1;
|
||||
}
|
||||
return nok(code);
|
||||
}
|
||||
effects.enter('mathTextData');
|
||||
return content(code);
|
||||
}
|
||||
|
||||
return regions;
|
||||
function content(code: Code): State | undefined {
|
||||
if (code === codes.eof || code === codes.graveAccent || markdownLineEnding(code)) {
|
||||
return nok(code);
|
||||
}
|
||||
if (code === codes.dollarSign) {
|
||||
if (markdownSpace(previousCode) || braceDepth !== 0) {
|
||||
return nok(code);
|
||||
}
|
||||
effects.exit('mathTextData');
|
||||
effects.enter('mathTextSequence');
|
||||
effects.consume(code);
|
||||
return close;
|
||||
}
|
||||
if (code === codes.backslash) {
|
||||
effects.consume(code);
|
||||
return escape;
|
||||
}
|
||||
if (code === codes.leftCurlyBrace) {
|
||||
braceDepth++;
|
||||
}
|
||||
if (code === codes.rightCurlyBrace) {
|
||||
if (braceDepth === 0) {
|
||||
return nok(code);
|
||||
}
|
||||
braceDepth--;
|
||||
}
|
||||
previousCode = code;
|
||||
effects.consume(code);
|
||||
return content;
|
||||
}
|
||||
|
||||
function escape(code: Code): State | undefined {
|
||||
if (code === codes.eof || markdownLineEnding(code)) {
|
||||
return nok(code);
|
||||
}
|
||||
previousCode = code;
|
||||
effects.consume(code);
|
||||
return content;
|
||||
}
|
||||
|
||||
function close(code: Code): State | undefined {
|
||||
if (code === codes.dollarSign || asciiDigit(code)) {
|
||||
return nok(code);
|
||||
}
|
||||
effects.exit('mathTextSequence');
|
||||
effects.exit('mathText');
|
||||
return ok(code);
|
||||
}
|
||||
|
||||
return start;
|
||||
}
|
||||
|
||||
/** Mirrors `micromark-extension-math`: a `$` opener is valid unless it follows an unescaped `$`. */
|
||||
function previous(this: TokenizeContext, code: Code): boolean {
|
||||
return (
|
||||
code !== codes.dollarSign ||
|
||||
this.events[this.events.length - 1][1].type === types.characterEscape
|
||||
);
|
||||
}
|
||||
|
||||
const mathSpan: Construct = {
|
||||
name: 'mathSpanSingleDollar',
|
||||
tokenize: tokenizeMathSpan,
|
||||
previous,
|
||||
};
|
||||
|
||||
/**
|
||||
* Checks if a position is inside any code block region using binary search
|
||||
* @param position The position to check
|
||||
* @param codeRegions Array of code block regions
|
||||
* @returns True if position is inside a code block
|
||||
* micromark syntax extension adding currency-safe single-dollar inline math. It emits the
|
||||
* same `mathText`/`mathTextData` tokens as `micromark-extension-math`, so `remark-math`'s
|
||||
* `mathFromMarkdown` handlers turn its spans into regular `inlineMath` nodes. Registration
|
||||
* order relative to `remark-math` is immaterial: this construct rejects `$$` openers and
|
||||
* `remark-math` (with `singleDollarTextMath: false`) rejects single `$` openers.
|
||||
*/
|
||||
function isInCodeBlock(position: number, codeRegions: Array<[number, number]>): boolean {
|
||||
let left = 0;
|
||||
let right = codeRegions.length - 1;
|
||||
|
||||
while (left <= right) {
|
||||
const mid = Math.floor((left + right) / 2);
|
||||
const [start, end] = codeRegions[mid];
|
||||
|
||||
if (position >= start && position <= end) {
|
||||
return true;
|
||||
} else if (position < start) {
|
||||
right = mid - 1;
|
||||
} else {
|
||||
left = mid + 1;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
export const singleDollarMath: Extension = {
|
||||
text: { [codes.dollarSign]: mathSpan },
|
||||
};
|
||||
|
||||
/**
|
||||
* Preprocesses LaTeX content by escaping currency indicators and converting single dollar math delimiters.
|
||||
* Optimized for high-frequency execution.
|
||||
* @param content The input string containing LaTeX expressions.
|
||||
* @returns The processed string with escaped currency indicators and converted math delimiters.
|
||||
* remark plugin enabling {@link singleDollarMath}. Must run alongside `remark-math`, which
|
||||
* registers the mdast handlers for the tokens this extension emits.
|
||||
*/
|
||||
export function preprocessLaTeX(content: string): string {
|
||||
// Early return for most common case
|
||||
if (!content.includes('$')) return content;
|
||||
|
||||
// Process mhchem first (usually rare, so check if needed)
|
||||
let processed = content;
|
||||
if (content.includes('\\ce{') || content.includes('\\pu{')) {
|
||||
processed = escapeMhchem(content);
|
||||
}
|
||||
|
||||
// Find all code block regions once
|
||||
const codeRegions = findCodeBlockRegions(processed);
|
||||
|
||||
// First pass: escape currency dollar signs
|
||||
const parts: string[] = [];
|
||||
let lastIndex = 0;
|
||||
|
||||
// Reset regex for reuse
|
||||
CURRENCY_REGEX.lastIndex = 0;
|
||||
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = CURRENCY_REGEX.exec(processed)) !== null) {
|
||||
if (!isInCodeBlock(match.index, codeRegions)) {
|
||||
parts.push(processed.substring(lastIndex, match.index));
|
||||
parts.push('\\$');
|
||||
lastIndex = match.index + 1;
|
||||
}
|
||||
}
|
||||
parts.push(processed.substring(lastIndex));
|
||||
processed = parts.join('');
|
||||
|
||||
// Second pass: convert single dollar delimiters to double dollars
|
||||
const result: string[] = [];
|
||||
lastIndex = 0;
|
||||
|
||||
// Reset regex for reuse
|
||||
SINGLE_DOLLAR_REGEX.lastIndex = 0;
|
||||
|
||||
while ((match = SINGLE_DOLLAR_REGEX.exec(processed)) !== null) {
|
||||
if (!isInCodeBlock(match.index, codeRegions)) {
|
||||
result.push(processed.substring(lastIndex, match.index));
|
||||
result.push(`$$${match[1]}$$`);
|
||||
lastIndex = match.index + match[0].length;
|
||||
}
|
||||
}
|
||||
result.push(processed.substring(lastIndex));
|
||||
|
||||
return result.join('');
|
||||
}
|
||||
export const remarkSingleDollarMath: Plugin<[], Root> = function remarkSingleDollarMath() {
|
||||
const data = this.data() as ParserData;
|
||||
const extensions = (data.micromarkExtensions ??= []);
|
||||
extensions.push(singleDollarMath);
|
||||
};
|
||||
|
|
|
|||
3
package-lock.json
generated
3
package-lock.json
generated
|
|
@ -986,6 +986,8 @@
|
|||
"micromark-extension-gfm": "^3.0.0",
|
||||
"micromark-extension-llm-math": "^3.1.0",
|
||||
"micromark-extension-math": "^3.1.0",
|
||||
"micromark-util-character": "^2.1.0",
|
||||
"micromark-util-symbol": "^2.0.0",
|
||||
"monaco-editor": "^0.56.0",
|
||||
"qrcode.react": "^4.2.0",
|
||||
"rc-input-number": "^7.4.2",
|
||||
|
|
@ -1054,6 +1056,7 @@
|
|||
"jest-environment-jsdom": "^30.2.0",
|
||||
"jest-file-loader": "^1.0.3",
|
||||
"jest-junit": "^17.0.0",
|
||||
"micromark-util-types": "^2.0.0",
|
||||
"postcss": "^8.5.18",
|
||||
"postcss-preset-env": "^11.2.0",
|
||||
"tailwindcss": "^3.4.1",
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue