Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .changeset/tidy-tokens-overlap.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,5 @@
---
'@livekit/components-core': patch
---

Fix `tokenize` returning duplicated text when several matches overlap a single longer match.
24 changes: 24 additions & 0 deletions packages/core/src/helper/tokenizer.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
import { describe, expect, it } from 'vitest';
import { tokenize } from './tokenizer';

describe('tokenize', () => {
it('splits text around matches', () => {
expect(tokenize('a foo b', { word: /foo/g })).toEqual([
'a ',
{ type: 'word', content: 'foo' },
' b',
]);
});

it('drops matches nested inside an earlier, longer match', () => {
const tokens = tokenize('abcdefghij', { outer: /abcdefghij/g, first: /cd/g, second: /gh/g });
expect(tokens).toEqual([{ type: 'outer', content: 'abcdefghij' }]);
});

it('does not duplicate text when several matches overlap one match', () => {
const input = 'xx abcdefghij yy';
const tokens = tokenize(input, { outer: /abcdefghij/g, first: /cd/g, second: /gh/g });
const rebuilt = tokens.map((t) => (typeof t === 'string' ? t : t.content)).join('');
expect(rebuilt).toBe(input);
});
});
10 changes: 6 additions & 4 deletions packages/core/src/helper/tokenizer.ts
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ export const createDefaultGrammar = () => {
};

export function tokenize<T extends TokenizeGrammar>(input: string, grammar: T) {
let lastEnd = 0;
const matches = Object.entries(grammar)
.map(([type, rx], weight) =>
Array.from(input.matchAll(rx)).map(({ index, 0: content }) => ({
Expand All @@ -25,10 +26,11 @@ export function tokenize<T extends TokenizeGrammar>(input: string, grammar: T) {
const d = a.index - b.index;
return d !== 0 ? d : a.weight - b.weight;
})
.filter(({ index }, i, arr) => {
if (i === 0) return true;
const prev = arr[i - 1];
return prev.index + prev.content.length <= index;
.filter(({ index, content }) => {
// Keep a match only if it starts after the end of the last kept match.
if (index < lastEnd) return false;
lastEnd = index + content.length;
return true;
});

const tokens = [];
Expand Down
Loading