diff --git a/.changeset/tidy-tokens-overlap.md b/.changeset/tidy-tokens-overlap.md new file mode 100644 index 000000000..9c0667fe2 --- /dev/null +++ b/.changeset/tidy-tokens-overlap.md @@ -0,0 +1,5 @@ +--- +'@livekit/components-core': patch +--- + +Fix `tokenize` returning duplicated text when several matches overlap a single longer match. diff --git a/packages/core/src/helper/tokenizer.test.ts b/packages/core/src/helper/tokenizer.test.ts new file mode 100644 index 000000000..75663f586 --- /dev/null +++ b/packages/core/src/helper/tokenizer.test.ts @@ -0,0 +1,24 @@ +import { describe, expect, it } from 'vitest'; +import { tokenize } from './tokenizer'; + +describe('tokenize', () => { + it('splits text around matches', () => { + expect(tokenize('a foo b', { word: /foo/g })).toEqual([ + 'a ', + { type: 'word', content: 'foo' }, + ' b', + ]); + }); + + it('drops matches nested inside an earlier, longer match', () => { + const tokens = tokenize('abcdefghij', { outer: /abcdefghij/g, first: /cd/g, second: /gh/g }); + expect(tokens).toEqual([{ type: 'outer', content: 'abcdefghij' }]); + }); + + it('does not duplicate text when several matches overlap one match', () => { + const input = 'xx abcdefghij yy'; + const tokens = tokenize(input, { outer: /abcdefghij/g, first: /cd/g, second: /gh/g }); + const rebuilt = tokens.map((t) => (typeof t === 'string' ? t : t.content)).join(''); + expect(rebuilt).toBe(input); + }); +}); diff --git a/packages/core/src/helper/tokenizer.ts b/packages/core/src/helper/tokenizer.ts index b187d025a..c67e6ebb7 100644 --- a/packages/core/src/helper/tokenizer.ts +++ b/packages/core/src/helper/tokenizer.ts @@ -11,6 +11,7 @@ export const createDefaultGrammar = () => { }; export function tokenize(input: string, grammar: T) { + let lastEnd = 0; const matches = Object.entries(grammar) .map(([type, rx], weight) => Array.from(input.matchAll(rx)).map(({ index, 0: content }) => ({ @@ -25,10 +26,11 @@ export function tokenize(input: string, grammar: T) { const d = a.index - b.index; return d !== 0 ? d : a.weight - b.weight; }) - .filter(({ index }, i, arr) => { - if (i === 0) return true; - const prev = arr[i - 1]; - return prev.index + prev.content.length <= index; + .filter(({ index, content }) => { + // Keep a match only if it starts after the end of the last kept match. + if (index < lastEnd) return false; + lastEnd = index + content.length; + return true; }); const tokens = [];