* perf: pre-compile KaTeX Unicode regex at module load time
The katexStart() function was creating a new RegExp with Unicode
property escapes (\p{Script=Han}, \p{Script=Hiragana}, etc.) on
every invocation. Unicode property escapes are extremely expensive
to compile as the regex engine must build character class tables
covering tens of thousands of code points.
Since marked calls the start() function at every character position
while scanning source text, this meant hundreds of regex compilations
per marked.lexer() call, and lexer runs ~60 times/sec during streaming.
Profiling showed KaTeX regex consuming 87% (320ms/365ms) of total
markdown rendering time.
Changes:
- Pre-compile SURROUNDING_CHARS_REGEX once at module load time
- Use .test() instead of .match() to avoid array allocations
- Fix delimiter search to find earliest match, not last match
* perf: replace katexStart with single-pass character scan
The katexStart() function was the dominant cost in marked's lexer,
consuming 55-58% of total markdown rendering time per profiling.
It was called at every character position by marked and each call:
- Looped through 3-5 delimiters, each doing indexOf() on the full
remaining source (3-5 x O(n) string scans per call)
- Ran the complex ruleReg regex with Unicode lookaheads for validation
- On failed validation, created substrings and looped again
Replace with a single linear character scan using charCodeAt that:
- Checks only for $ (charCode 36) or backslash (charCode 92)
- Filters backslash hits by next character to avoid false positives
- Preserves the surrounding-character validation
- Returns immediately on first valid candidate
- Lets the tokenizer handle full validation (it already does this)
This reduces start() from O(n * delimiters * retries) to O(n) with
a very small constant factor per call.
* Update katex-extension.ts
156 lines
4.6 KiB
TypeScript
156 lines
4.6 KiB
TypeScript
const DELIMITER_LIST = [
|
|
{ left: '$$', right: '$$', display: true },
|
|
{ left: '$', right: '$', display: false },
|
|
{ left: '\\pu{', right: '}', display: false },
|
|
{ left: '\\ce{', right: '}', display: false },
|
|
{ left: '\\(', right: '\\)', display: false },
|
|
{ left: '\\[', right: '\\]', display: true },
|
|
{ left: '\\begin{equation}', right: '\\end{equation}', display: true }
|
|
];
|
|
|
|
// Defines characters that are allowed to immediately precede or follow a math delimiter.
|
|
const ALLOWED_SURROUNDING_CHARS =
|
|
'\\s。,、、;;„“‘’“”()「」『』[]《》【】‹›«»…⋯::?!~⇒?!-\\/:-@\\[-`{-~\\p{Script=Han}\\p{Script=Hiragana}\\p{Script=Katakana}\\p{Script=Hangul}';
|
|
// Modified to fit more formats in different languages. Originally: '\\s?。,、;!-\\/:-@\\[-`{-~\\p{Script=Han}\\p{Script=Hiragana}\\p{Script=Katakana}\\p{Script=Hangul}';
|
|
|
|
// Pre-compile the surrounding character regex once at module load time.
|
|
// This regex uses Unicode property escapes (\p{Script=Han}, etc.) which are
|
|
// extremely expensive to compile - doing so on every call caused ~87% of
|
|
// markdown rendering time to be spent in KaTeX regex compilation.
|
|
const ALLOWED_SURROUNDING_CHARS_REGEX = new RegExp(`[${ALLOWED_SURROUNDING_CHARS}]`, 'u');
|
|
|
|
// const DELIMITER_LIST = [
|
|
// { left: '$$', right: '$$', display: false },
|
|
// { left: '$', right: '$', display: false },
|
|
// ];
|
|
|
|
// const inlineRule = /^(\${1,2})(?!\$)((?:\\.|[^\\\n])*?(?:\\.|[^\\\n\$]))\1(?=[\s?!\.,:?!。,:]|$)/;
|
|
// const blockRule = /^(\${1,2})\n((?:\\[^]|[^\\])+?)\n\1(?:\n|$)/;
|
|
|
|
const inlinePatterns = [];
|
|
const blockPatterns = [];
|
|
|
|
function escapeRegex(string) {
|
|
return string.replace(/[-\/\\^$*+?.()|[\]{}]/g, '\\$&');
|
|
}
|
|
|
|
function generateRegexRules(delimiters) {
|
|
delimiters.forEach((delimiter) => {
|
|
const { left, right, display } = delimiter;
|
|
// Ensure regex-safe delimiters
|
|
const escapedLeft = escapeRegex(left);
|
|
const escapedRight = escapeRegex(right);
|
|
|
|
if (!display) {
|
|
// For inline delimiters, we match everything
|
|
inlinePatterns.push(`${escapedLeft}((?:\\\\[^]|[^\\\\])+?)${escapedRight}`);
|
|
} else {
|
|
// Block delimiters doubles as inline delimiters when not followed by a newline
|
|
inlinePatterns.push(`${escapedLeft}(?!\\n)((?:\\\\[^]|[^\\\\])+?)(?!\\n)${escapedRight}`);
|
|
blockPatterns.push(`${escapedLeft}\\n((?:\\\\[^]|[^\\\\])+?)\\n${escapedRight}`);
|
|
}
|
|
});
|
|
|
|
// Math formulas can end in special characters
|
|
const inlineRule = new RegExp(
|
|
`^(${inlinePatterns.join('|')})(?=[${ALLOWED_SURROUNDING_CHARS}]|$)`,
|
|
'u'
|
|
);
|
|
const blockRule = new RegExp(
|
|
`^(${blockPatterns.join('|')})(?=[${ALLOWED_SURROUNDING_CHARS}]|$)`,
|
|
'u'
|
|
);
|
|
|
|
return { inlineRule, blockRule };
|
|
}
|
|
|
|
const { inlineRule, blockRule } = generateRegexRules(DELIMITER_LIST);
|
|
|
|
export default function (options = {}) {
|
|
return {
|
|
extensions: [inlineKatex(options), blockKatex(options)]
|
|
};
|
|
}
|
|
|
|
function katexStart(src, displayMode: boolean) {
|
|
for (let i = 0; i < src.length; i++) {
|
|
const ch = src.charCodeAt(i);
|
|
|
|
if (ch === 36 /* $ */) {
|
|
// Display mode requires $$, skip single $ for display
|
|
if (displayMode && src.charAt(i + 1) !== '$') {
|
|
continue;
|
|
}
|
|
if (i === 0 || ALLOWED_SURROUNDING_CHARS_REGEX.test(src.charAt(i - 1))) {
|
|
return i;
|
|
}
|
|
} else if (ch === 92 /* \ */) {
|
|
const next = src.charAt(i + 1);
|
|
// Only consider \ if followed by a valid math delimiter start
|
|
if (displayMode) {
|
|
// Display: \[ or \begin{equation}
|
|
if (next !== '[' && next !== 'b') continue;
|
|
} else {
|
|
// Inline: \( or \ce{ or \pu{
|
|
if (next !== '(' && next !== 'c' && next !== 'p') continue;
|
|
}
|
|
if (i === 0 || ALLOWED_SURROUNDING_CHARS_REGEX.test(src.charAt(i - 1))) {
|
|
return i;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
function katexTokenizer(src, tokens, displayMode: boolean) {
|
|
const ruleReg = displayMode ? blockRule : inlineRule;
|
|
const type = displayMode ? 'blockKatex' : 'inlineKatex';
|
|
|
|
const match = src.match(ruleReg);
|
|
|
|
if (match) {
|
|
const text = match
|
|
.slice(2)
|
|
.filter((item) => item)
|
|
.find((item) => item.trim());
|
|
|
|
return {
|
|
type,
|
|
raw: match[0],
|
|
text: text,
|
|
displayMode
|
|
};
|
|
}
|
|
}
|
|
|
|
function inlineKatex(options) {
|
|
return {
|
|
name: 'inlineKatex',
|
|
level: 'inline',
|
|
start(src) {
|
|
return katexStart(src, false);
|
|
},
|
|
tokenizer(src, tokens) {
|
|
return katexTokenizer(src, tokens, false);
|
|
},
|
|
renderer(token) {
|
|
return `${token?.text ?? ''}`;
|
|
}
|
|
};
|
|
}
|
|
|
|
function blockKatex(options) {
|
|
return {
|
|
name: 'blockKatex',
|
|
level: 'block',
|
|
start(src) {
|
|
return katexStart(src, true);
|
|
},
|
|
tokenizer(src, tokens) {
|
|
return katexTokenizer(src, tokens, true);
|
|
},
|
|
renderer(token) {
|
|
return `${token?.text ?? ''}`;
|
|
}
|
|
};
|
|
}
|