From 5ee126ff15bf04d9890594adfc312d41c8e6bd71 Mon Sep 17 00:00:00 2001 From: David <244863912+dyk1454683243-sudo@users.noreply.github.com> Date: Sun, 20 Sep 2026 03:00:36 +0000 Subject: [PATCH] fix: avoid quadratic scan for unmatched emphasis delimiters Tokenizer.emStrong and del walked every later delimiter for each unmatched opener, so parse time grew ~4x when those inputs doubled. Cache the last possible closer on maskedSrc and skip the scan when none remains after the opener. Fixes #4099 Co-authored-by: David --- src/Tokenizer.ts | 95 +++++++++++++++++-- .../redos/quadratic_unmatched_emphasis.cjs | 21 ++++ test/unit/marked.test.js | 33 +++++++ 3 files changed, 143 insertions(+), 6 deletions(-) create mode 100644 test/specs/redos/quadratic_unmatched_emphasis.cjs diff --git a/src/Tokenizer.ts b/src/Tokenizer.ts index b5385ce620..edf4f1552d 100644 --- a/src/Tokenizer.ts +++ b/src/Tokenizer.ts @@ -113,6 +113,84 @@ function isLabelEndInsideToken(src: string, label: string, labelStart: number, r return false; } +type CloserDelim = '*' | '_' | '~'; + +interface CloserHints { + '*': number; + _: number; + '~': number; +} + +let closerHintSrc = ''; +let closerHints: CloserHints | undefined; + +function isCloserDelim(ch: string): ch is CloserDelim { + return ch === '*' || ch === '_' || ch === '~'; +} + +/** + * Last index of a `*`, `_`, or `~` run that can still close — i.e. not + * preceded by ASCII whitespace or the start of the string. + * + * The right-delim regexes only treat a run as a closer in groups 1, 2, 5 + * and 6, all of which require a non-whitespace predecessor. If nothing + * after `windowStart` qualifies, the existing scan cannot succeed, so + * unmatched openers can skip it. A hit only means "keep looking": + * CommonMark rules 9–10 and mid-run openers are still applied in the loop. + * + * Cached per maskedSrc so each opener in an inline run does not walk the + * remainder again (#4099). + */ +function lastPossibleCloserIndex(src: string, delim: CloserDelim): number { + if (closerHintSrc !== src || !closerHints) { + closerHintSrc = src; + closerHints = scanLastPossibleClosers(src); + } + return closerHints[delim]; +} + +function scanLastPossibleClosers(src: string): CloserHints { + const last: CloserHints = { '*': -1, _: -1, '~': -1 }; + const len = src.length; + for (let i = 0; i < len; i++) { + const ch = src[i]; + if (!isCloserDelim(ch)) { + continue; + } + const prev = i === 0 ? '\n' : src[i - 1]; + let j = i + 1; + while (j < len && src[j] === ch) { + j++; + } + if (prev !== ' ' && prev !== '\t' && prev !== '\n' && prev !== '\r' && prev !== '\f') { + last[ch] = i; + } + i = j - 1; + } + return last; +} + +interface DelimWindow { + maskedSrc: string; + lastCloser: number; +} + +/** + * Clip maskedSrc to the same window `emStrong` / `del` already scan, or + * return undefined when that window has no possible closer. + */ +function remainingDelimWindow(maskedSrc: string, srcLength: number, lLength: number, delim: CloserDelim): DelimWindow | undefined { + const windowStart = maskedSrc.length - srcLength + lLength; + const lastCloser = lastPossibleCloserIndex(maskedSrc, delim); + if (lastCloser < windowStart) { + return; + } + return { + maskedSrc: maskedSrc.slice(-1 * srcLength + lLength), + lastCloser: lastCloser - windowStart, + }; +} + /** * Tokenizer */ @@ -823,17 +901,20 @@ export class _Tokenizer { let rDelim, rLength, delimTotal = lLength, midDelimTotal = 0; const delimChar = match[0][0]; + if (!isCloserDelim(delimChar)) return; // A mid-run opener (for example the second star of an unmatched `**`) must // only pair with a delimiter that can only close, otherwise it steals the // opener of a later span (`**a*b*c` must be `**abc`). const midRun = prevChar === delimChar; const endReg = delimChar === '*' ? this.rules.inline.emStrongRDelimAst : this.rules.inline.emStrongRDelimUnd; - endReg.lastIndex = 0; - + const window = remainingDelimWindow(maskedSrc, src.length, lLength, delimChar); + if (!window) return; // Clip maskedSrc to same section of string as src (move to lexer?) - maskedSrc = maskedSrc.slice(-1 * src.length + lLength); + maskedSrc = window.maskedSrc; + endReg.lastIndex = 0; while ((match = endReg.exec(maskedSrc)) !== null) { + if (match.index > window.lastCloser) break; rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6]; if (!rDelim) continue; // skip single * in __abc*abc__ @@ -927,12 +1008,14 @@ export class _Tokenizer { let rDelim, rLength, delimTotal = lLength; const endReg = this.rules.inline.delRDelim; - endReg.lastIndex = 0; - + const window = remainingDelimWindow(maskedSrc, src.length, lLength, '~'); + if (!window) return; // Clip maskedSrc to same section of string as src - maskedSrc = maskedSrc.slice(-1 * src.length + lLength); + maskedSrc = window.maskedSrc; + endReg.lastIndex = 0; while ((match = endReg.exec(maskedSrc)) !== null) { + if (match.index > window.lastCloser) break; rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6]; if (!rDelim) continue; diff --git a/test/specs/redos/quadratic_unmatched_emphasis.cjs b/test/specs/redos/quadratic_unmatched_emphasis.cjs new file mode 100644 index 0000000000..bd2af931ee --- /dev/null +++ b/test/specs/redos/quadratic_unmatched_emphasis.cjs @@ -0,0 +1,21 @@ +// Unmatched left-flanking emphasis / strikethrough delimiters. Before the +// last-closer early-exit, each opener walked every later delimiter +// (issue #4099). These sizes exceed the spec harness 1s budget on that path. +module.exports = [ + { + markdown: ('- *').repeat(2000), + html: `
    \n
  • ${'*- '.repeat(1999)}*
  • \n
\n`, + }, + { + markdown: ('+ _').repeat(2000), + html: `
    \n
  • ${'_+ '.repeat(1999)}_
  • \n
\n`, + }, + { + markdown: ('*x *x ').repeat(2000), + html: `

${'*x *x '.repeat(2000)}

\n`, + }, + { + markdown: ('~x ~x ').repeat(2000), + html: `

${'~x ~x '.repeat(2000)}

\n`, + }, +]; diff --git a/test/unit/marked.test.js b/test/unit/marked.test.js index 5d1449b4aa..676c9eca39 100644 --- a/test/unit/marked.test.js +++ b/test/unit/marked.test.js @@ -1122,4 +1122,37 @@ br assert.strictEqual(html.trim(), '

text

'); }); }); + + describe('unmatched emphasis delimiters', () => { + it('should parse unmatched * openers in well under a second', () => { + const src = ('- *').repeat(2000); + const start = performance.now(); + const html = marked.parse(src); + const ms = performance.now() - start; + assert.ok(ms < 1000, `expected linear parse, took ${Math.round(ms)}ms`); + assert.strictEqual(html, `
    \n
  • ${'*- '.repeat(1999)}*
  • \n
\n`); + }); + + it('should parse unmatched _ and ~ openers quickly', () => { + const start = performance.now(); + assert.strictEqual( + marked.parse(('+ _').repeat(2000)), + `
    \n
  • ${'_+ '.repeat(1999)}_
  • \n
\n`, + ); + assert.strictEqual( + marked.parse(('~x ~x ').repeat(2000)), + `

${'~x ~x '.repeat(2000)}

\n`, + ); + const ms = performance.now() - start; + assert.ok(ms < 1000, `expected linear parse, took ${Math.round(ms)}ms`); + }); + + it('should keep mid-run openers and emphasis rules 9-10', () => { + assert.strictEqual(marked.parse('**a*b*c').trim(), '

**abc

'); + assert.strictEqual(marked.parse('*foo**bar*').trim(), '

foo**bar

'); + assert.strictEqual(marked.parse('*foo *bar*').trim(), '

*foo bar

'); + assert.strictEqual(marked.parse('~~foo~~').trim(), '

foo

'); + assert.strictEqual(marked.parse('~~foo~').trim(), '

~~foo~

'); + }); + }); });