Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
95 changes: 89 additions & 6 deletions src/Tokenizer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -113,6 +113,84 @@ function isLabelEndInsideToken(src: string, label: string, labelStart: number, r
return false;
}

type CloserDelim = '*' | '_' | '~';

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Can we be more descriptive. closer of what?


interface CloserHints {
'*': number;
_: number;
'~': number;
}

let closerHintSrc = '';
let closerHints: CloserHints | undefined;
Comment on lines +124 to +125

function isCloserDelim(ch: string): ch is CloserDelim {
return ch === '*' || ch === '_' || ch === '~';
}

/**
* Last index of a `*`, `_`, or `~` run that can still close — i.e. not
* preceded by ASCII whitespace or the start of the string.
*
* The right-delim regexes only treat a run as a closer in groups 1, 2, 5
* and 6, all of which require a non-whitespace predecessor. If nothing
* after `windowStart` qualifies, the existing scan cannot succeed, so
* unmatched openers can skip it. A hit only means "keep looking":
* CommonMark rules 9–10 and mid-run openers are still applied in the loop.
*
* Cached per maskedSrc so each opener in an inline run does not walk the
* remainder again (#4099).
*/
function lastPossibleCloserIndex(src: string, delim: CloserDelim): number {
if (closerHintSrc !== src || !closerHints) {
closerHintSrc = src;
closerHints = scanLastPossibleClosers(src);
}
return closerHints[delim];
}

function scanLastPossibleClosers(src: string): CloserHints {
const last: CloserHints = { '*': -1, _: -1, '~': -1 };
const len = src.length;
for (let i = 0; i < len; i++) {
const ch = src[i];
if (!isCloserDelim(ch)) {
continue;
}
const prev = i === 0 ? '\n' : src[i - 1];
let j = i + 1;
while (j < len && src[j] === ch) {
j++;
}
if (prev !== ' ' && prev !== '\t' && prev !== '\n' && prev !== '\r' && prev !== '\f') {
last[ch] = i;
}
i = j - 1;
}
return last;
}

interface DelimWindow {
maskedSrc: string;
lastCloser: number;
}

/**
* Clip maskedSrc to the same window `emStrong` / `del` already scan, or
* return undefined when that window has no possible closer.
*/
function remainingDelimWindow(maskedSrc: string, srcLength: number, lLength: number, delim: CloserDelim): DelimWindow | undefined {
const windowStart = maskedSrc.length - srcLength + lLength;
const lastCloser = lastPossibleCloserIndex(maskedSrc, delim);
if (lastCloser < windowStart) {
return;
}
return {
maskedSrc: maskedSrc.slice(-1 * srcLength + lLength),
lastCloser: lastCloser - windowStart,
};
}

/**
* Tokenizer
*/
Expand Down Expand Up @@ -823,17 +901,20 @@ export class _Tokenizer<ParserOutput = string, RendererOutput = string> {
let rDelim, rLength, delimTotal = lLength, midDelimTotal = 0;

const delimChar = match[0][0];
if (!isCloserDelim(delimChar)) return;
// A mid-run opener (for example the second star of an unmatched `**`) must
// only pair with a delimiter that can only close, otherwise it steals the
// opener of a later span (`**a*b*c` must be `**a<em>b</em>c`).
const midRun = prevChar === delimChar;
const endReg = delimChar === '*' ? this.rules.inline.emStrongRDelimAst : this.rules.inline.emStrongRDelimUnd;
endReg.lastIndex = 0;

const window = remainingDelimWindow(maskedSrc, src.length, lLength, delimChar);
if (!window) return;
// Clip maskedSrc to same section of string as src (move to lexer?)
maskedSrc = maskedSrc.slice(-1 * src.length + lLength);
maskedSrc = window.maskedSrc;
endReg.lastIndex = 0;

while ((match = endReg.exec(maskedSrc)) !== null) {
if (match.index > window.lastCloser) break;
rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6];

if (!rDelim) continue; // skip single * in __abc*abc__
Expand Down Expand Up @@ -927,12 +1008,14 @@ export class _Tokenizer<ParserOutput = string, RendererOutput = string> {
let rDelim, rLength, delimTotal = lLength;

const endReg = this.rules.inline.delRDelim;
endReg.lastIndex = 0;

const window = remainingDelimWindow(maskedSrc, src.length, lLength, '~');
if (!window) return;
// Clip maskedSrc to same section of string as src
maskedSrc = maskedSrc.slice(-1 * src.length + lLength);
maskedSrc = window.maskedSrc;
endReg.lastIndex = 0;

while ((match = endReg.exec(maskedSrc)) !== null) {
if (match.index > window.lastCloser) break;
rDelim = match[1] || match[2] || match[3] || match[4] || match[5] || match[6];

if (!rDelim) continue;
Expand Down
21 changes: 21 additions & 0 deletions test/specs/redos/quadratic_unmatched_emphasis.cjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
// Unmatched left-flanking emphasis / strikethrough delimiters. Before the
// last-closer early-exit, each opener walked every later delimiter
// (issue #4099). These sizes exceed the spec harness 1s budget on that path.
module.exports = [
{
markdown: ('- *').repeat(2000),
html: `<ul>\n<li>${'*- '.repeat(1999)}*</li>\n</ul>\n`,
},
{
markdown: ('+ _').repeat(2000),
html: `<ul>\n<li>${'_+ '.repeat(1999)}_</li>\n</ul>\n`,
},
{
markdown: ('*x *x ').repeat(2000),
html: `<p>${'*x *x '.repeat(2000)}</p>\n`,
},
{
markdown: ('~x ~x ').repeat(2000),
html: `<p>${'~x ~x '.repeat(2000)}</p>\n`,
},
];
33 changes: 33 additions & 0 deletions test/unit/marked.test.js
Original file line number Diff line number Diff line change
Expand Up @@ -1122,4 +1122,37 @@ br
assert.strictEqual(html.trim(), '<p><em>text</em></p>');
});
});

describe('unmatched emphasis delimiters', () => {

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I don't think we need the performance tests if we have the redos tests. And the mid-run opener tests should be in test/specs/new

it('should parse unmatched * openers in well under a second', () => {
const src = ('- *').repeat(2000);
const start = performance.now();
const html = marked.parse(src);
const ms = performance.now() - start;
assert.ok(ms < 1000, `expected linear parse, took ${Math.round(ms)}ms`);
assert.strictEqual(html, `<ul>\n<li>${'*- '.repeat(1999)}*</li>\n</ul>\n`);
});

it('should parse unmatched _ and ~ openers quickly', () => {
const start = performance.now();
assert.strictEqual(
marked.parse(('+ _').repeat(2000)),
`<ul>\n<li>${'_+ '.repeat(1999)}_</li>\n</ul>\n`,
);
assert.strictEqual(
marked.parse(('~x ~x ').repeat(2000)),
`<p>${'~x ~x '.repeat(2000)}</p>\n`,
);
const ms = performance.now() - start;
assert.ok(ms < 1000, `expected linear parse, took ${Math.round(ms)}ms`);
});

it('should keep mid-run openers and emphasis rules 9-10', () => {
assert.strictEqual(marked.parse('**a*b*c').trim(), '<p>**a<em>b</em>c</p>');
assert.strictEqual(marked.parse('*foo**bar*').trim(), '<p><em>foo**bar</em></p>');
assert.strictEqual(marked.parse('*foo *bar*').trim(), '<p>*foo <em>bar</em></p>');
assert.strictEqual(marked.parse('~~foo~~').trim(), '<p><del>foo</del></p>');
assert.strictEqual(marked.parse('~~foo~').trim(), '<p>~~foo~</p>');
});
});
});
Loading