diff --git a/CHANGELOG.md b/CHANGELOG.md index 9ce80ac..44816a3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -48,7 +48,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - New constants `ZVec::INDEX_TYPE_FTS`, `ZVec::QUERY_PARAM_FTS` (both `11`) plus `FTS_OPERATOR_OR` / `FTS_OPERATOR_AND`. - FFI: `zvec_index_params_set_fts()`, `zvec_vector_query_set_fts()`, and `IndexType::FTS` handling in the params factory and `to_index_type()`. - Test: `tests/test_fts.phpt` (index creation, OR/AND semantics, lowercase folding, no-match, input validation). - - Hybrid dense + FTS retrieval through `MultiQuery`, and stemming filters beyond the tested defaults, are not covered yet. + - The `ngram` and `jieba` tokenizers, the `stemmer` filter, and their `extraParams` keys are now covered by `tests/test_fts_tokenizer_ngram.phpt`, `tests/test_fts_tokenizer_jieba.phpt` and `tests/test_fts_filter_stemmer.phpt`. The `forFts()` docblock listed a `stemmer_en` filter and a non-JSON `stemmer_lang=en` example, neither of which upstream accepts; both are now documented correctly and asserted as rejected. + - Hybrid dense + FTS retrieval through `MultiQuery` is not covered yet. - **DiskANN index type and query params** (#179) - `ZVecIndexParams::forDiskAnn(metricType, maxDegree, listSize, pqChunkNum, quantizeType)` — the disk-based graph index, previously missing because only the in-memory Vamana variant was exposed. Mirrors the official Go SDK `NewDiskANNIndexParams`. diff --git a/README.md b/README.md index a2be5ac..d247930 100644 --- a/README.md +++ b/README.md @@ -522,6 +522,10 @@ $params = ZVecIndexParams::forFts( // The jieba dictionary ships in zvec_data/jieba_dict next to the library and is // used automatically. Lookup order: per-field extraParams jieba_dict_dir, then // ZVEC_JIEBA_DICT_DIR, then ZVec::init(jiebaDictDir:), then the bundled copy. +// +// A bad jieba dictionary path is not a catchable error: cppjieba calls abort(), +// which kills the process with exit code 134. The bundled copy avoids this, and +// ZVec::init(jiebaDictDir:) validates that both dictionary files exist. // Invert — keyword-based inverted index $params = ZVecIndexParams::forInvert( diff --git a/src/ZVecIndexParams.php b/src/ZVecIndexParams.php index 27a33d5..8c9e67a 100644 --- a/src/ZVecIndexParams.php +++ b/src/ZVecIndexParams.php @@ -199,20 +199,16 @@ public static function forDiskAnn(int $metricType, int $maxDegree = 100, int $li return new self($handle); } - /** - * Create Invert index params - * - * @throws ZVecException On FFI error - */ /** * Create Full-Text Search index params * * Inverted index over a STRING column. Mirrors the official Go SDK * NewFTSIndexParams(tokenizerName, filters, extraParams). * - * @param string $tokenizer Tokenizer name ("standard", …) - * @param string[] $filters Token filters, e.g. ["lowercase", "stemmer_en"] - * @param string $extraParams Extra params, e.g. "stemmer_lang=en" + * @param string $tokenizer "standard", "ngram", "jieba" or "whitespace" + * @param string[] $filters Any of "lowercase", "ascii_folding", "stemmer" + * @param string $extraParams JSON object, e.g. '{"stemmer_lang":"english"}', + * '{"ngram_min":2,"ngram_max":3}', '{"cut_mode":"mix"}' * * @throws ZVecException On FFI error */ diff --git a/tests/test_fts_filter_stemmer.phpt b/tests/test_fts_filter_stemmer.phpt new file mode 100644 index 0000000..f52f506 --- /dev/null +++ b/tests/test_fts_filter_stemmer.phpt @@ -0,0 +1,140 @@ +--TEST-- +FTS filter "stemmer": Snowball stemming, stemmer_lang, and rejected filter names +--SKIPIF-- + +--FILE-- + $docs + * @return array{0: ZVec, 1: string} collection and its path + */ +function makeCollection(string $name, array $docs, array $filters, string $extraParams = ''): array +{ + $path = __DIR__ . '/../test_dbs/' . $name . '_' . uniqid(); + $schema = new ZVecSchema($name); + $schema->addString('body') + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $c = ZVec::create($path, $schema); + $c->createIndex('body', ZVecIndexParams::forFts( + tokenizer: 'standard', + filters: $filters, + extraParams: $extraParams, + )); + foreach ($docs as $pk => $body) { + $c->insert( + (new ZVecDoc($pk))->setString('body', $body)->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0]) + ); + } + $c->flush(); + return [$c, $path]; +} + +/** @return string sorted PKs; result order is not part of the contract */ +function ftsPks(ZVec $c, string $query): string +{ + $q = (new ZVecVectorQuery('body', []))->setTopk(10)->setFts('body', $query); + $pks = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $c->queryVector($q)); + sort($pks); + return $pks === [] ? '(none)' : implode(',', $pks); +} + +$docs = [ + 'd1' => 'the quick brown fox jumps', + 'd2' => 'foxes are clever animals', + 'd3' => 'running runners run daily', + 'd4' => 'completely unrelated text', +]; + +// Without stemming each form is indexed literally, so "fox" and "foxes" are +// different terms. This is the baseline the stemmer below is measured against. +[$c, $p] = makeCollection('fts_stem_base', $docs, ['lowercase']); +try { + echo 'base fox: ' . ftsPks($c, 'fox') . "\n"; + echo 'base foxes: ' . ftsPks($c, 'foxes') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +// Snowball english is the default language. +[$c, $p] = makeCollection('fts_stem_en', $docs, ['lowercase', 'stemmer']); +try { + echo 'fox: ' . ftsPks($c, 'fox') . "\n"; + echo 'foxes: ' . ftsPks($c, 'foxes') . "\n"; + echo 'run: ' . ftsPks($c, 'run') . "\n"; + echo 'running: ' . ftsPks($c, 'running') . "\n"; + + $c->close(); + $c = ZVec::open($p); + echo 'fox after reopen: ' . ftsPks($c, 'fox') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +[$c, $p] = makeCollection('fts_stem_porter', $docs, ['lowercase', 'stemmer'], '{"stemmer_lang":"porter"}'); +try { + echo 'porter fox: ' . ftsPks($c, 'fox') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +// A different Snowball language, with German umlauts in the documents. +$german = ['d1' => 'Häuser und Katzen', 'd2' => 'Haus']; +[$c, $p] = makeCollection('fts_stem_de', $german, ['lowercase', 'stemmer'], '{"stemmer_lang":"german"}'); +try { + echo 'german haus: ' . ftsPks($c, 'haus') . "\n"; + echo 'german katze: ' . ftsPks($c, 'katze') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +// The two bad examples from the old forFts() docblock, which showed values that +// upstream rejects. Kept here so the docblock fix is backed by a test. +$errors = [ + 'unknown lang' => [['lowercase', 'stemmer'], '{"stemmer_lang":"klingon"}'], + 'stemmer_en' => [['lowercase', 'stemmer_en'], ''], + 'not json' => [['lowercase'], 'stemmer_lang=en'], + 'bad tokenizer' => [['lowercase'], ''], +]; +foreach ($errors as $label => [$filters, $extra]) { + $path = __DIR__ . '/../test_dbs/fts_stem_err_' . uniqid(); + try { + $schema = new ZVecSchema('fts_stem_err'); + $schema->addString('body') + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $c = ZVec::create($path, $schema); + try { + $c->createIndex('body', ZVecIndexParams::forFts( + tokenizer: $label === 'bad tokenizer' ? 'bogus' : 'standard', + filters: $filters, + extraParams: $extra, + )); + echo "$label: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo "$label: rejected\n"; + } + } finally { + exec('rm -rf ' . escapeshellarg($path)); + } +} +?> +--EXPECT-- +base fox: d1 +base foxes: d2 +fox: d1,d2 +foxes: d1,d2 +run: d3 +running: d3 +fox after reopen: d1,d2 +porter fox: d1,d2 +german haus: d1,d2 +german katze: d1 +unknown lang: rejected +stemmer_en: rejected +not json: rejected +bad tokenizer: rejected diff --git a/tests/test_fts_tokenizer_jieba.phpt b/tests/test_fts_tokenizer_jieba.phpt new file mode 100644 index 0000000..1e807ed --- /dev/null +++ b/tests/test_fts_tokenizer_jieba.phpt @@ -0,0 +1,182 @@ +--TEST-- +FTS tokenizer "jieba": bundled dictionary, cut modes, user dictionary, and errors +--SKIPIF-- + +--FILE-- + $docs + * @return array{0: ZVec, 1: string} collection and its path + */ +function makeCollection(string $name, array $docs, array $filters, string $extraParams = ''): array +{ + $path = __DIR__ . '/../test_dbs/' . $name . '_' . uniqid(); + $schema = new ZVecSchema($name); + $schema->addString('body') + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $c = ZVec::create($path, $schema); + $c->createIndex('body', ZVecIndexParams::forFts( + tokenizer: 'jieba', + filters: $filters, + extraParams: $extraParams, + )); + foreach ($docs as $pk => $body) { + $c->insert( + (new ZVecDoc($pk))->setString('body', $body)->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0]) + ); + } + $c->flush(); + return [$c, $path]; +} + +/** @return string sorted PKs; result order is not part of the contract */ +function ftsPks(ZVec $c, string $query, ?string $matchString = null): string +{ + $q = (new ZVecVectorQuery('body', []))->setTopk(10) + ->setFts('body', $query, $matchString ?? ''); + $pks = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $c->queryVector($q)); + sort($pks); + return $pks === [] ? '(none)' : implode(',', $pks); +} + +$docs = [ + 'j1' => '我来到北京清华大学', + 'j2' => '他来到了网易杭研大厦', + 'j3' => '小明硕士毕业于中国科学院计算所', +]; + +// Default cut_mode is "search", which emits both short and long terms. +[$c, $p] = makeCollection('fts_jieba_default', $docs, []); +try { + echo '北京: ' . ftsPks($c, '北京') . "\n"; + echo '清华: ' . ftsPks($c, '清华') . "\n"; + echo '杭研: ' . ftsPks($c, '杭研') . "\n"; + echo '中国科学院: ' . ftsPks($c, '中国科学院') . "\n"; + echo '上海: ' . ftsPks($c, '上海') . "\n"; + // A phrase query goes through matchString rather than queryString. + echo 'phrase 北京大学: ' . ftsPks($c, '', '北京大学') . "\n"; + + $c->close(); + $c = ZVec::open($p); + echo '北京 after reopen: ' . ftsPks($c, '北京') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +// The cut mode really changes the tokens: "mix" favours precise single words and +// drops the compounds that "search" and "full" both emit. +$modes = ['search', 'full', 'mix', 'hmm']; +$terms = ['北京', '清华', '清华大学', '科学院', '大学']; +foreach ($modes as $mode) { + [$c, $p] = makeCollection('fts_jieba_' . $mode, $docs, [], json_encode(['cut_mode' => $mode])); + try { + foreach ($terms as $term) { + echo "$mode $term: " . ftsPks($c, $term) . "\n"; + } + } finally { + exec('rm -rf ' . escapeshellarg($p)); + } +} + +// Mixed Chinese/ASCII text, where lowercase still applies to the ASCII part. +[$c, $p] = makeCollection('fts_jieba_mixed', ['j1' => 'Hello World 你好世界'], ['lowercase']); +try { + echo 'mixed hello: ' . ftsPks($c, 'hello') . "\n"; + echo 'mixed 世界: ' . ftsPks($c, '世界') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +// A user dictionary adds a term. In "mix" mode, the added whole term then +// matches and the original sub-words stop matching, which is what makes the +// effect observable. +$dictFile = __DIR__ . '/../test_dbs/jieba_user_' . uniqid() . '.dict'; +file_put_contents($dictFile, "蓝鲸矢量库 1000 n\n"); +[$c, $p] = makeCollection('fts_jieba_user', ['u1' => '蓝鲸矢量库很快'], [], json_encode(['cut_mode' => 'mix'])); +try { + echo 'user 矢量: ' . ftsPks($c, '矢量') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} +[$c, $p] = makeCollection('fts_jieba_user2', ['u1' => '蓝鲸矢量库很快'], [], json_encode( + ['cut_mode' => 'mix', 'user_dict_path' => $dictFile], + JSON_UNESCAPED_SLASHES, +)); +try { + echo 'dict 矢量: ' . ftsPks($c, '矢量') . "\n"; + echo 'dict 蓝鲸矢量库: ' . ftsPks($c, '蓝鲸矢量库') . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); + unlink($dictFile); +} + +$path = __DIR__ . '/../test_dbs/fts_jieba_err_' . uniqid(); +try { + $schema = new ZVecSchema('fts_jieba_err'); + $schema->addString('body') + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $c = ZVec::create($path, $schema); + try { + $c->createIndex('body', ZVecIndexParams::forFts( + tokenizer: 'jieba', + filters: [], + extraParams: '{"cut_mode":"bogus"}', + )); + echo "bad cut_mode: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo 'bad cut_mode: ' . (str_contains($e->getMessage(), "unknown cut_mode 'bogus'") ? 'rejected' : $e->getMessage()) . "\n"; + } +} finally { + exec('rm -rf ' . escapeshellarg($path)); +} +?> +--EXPECT-- +北京: j1 +清华: j1 +杭研: j2 +中国科学院: j3 +上海: (none) +phrase 北京大学: j1 +北京 after reopen: j1 +search 北京: j1 +search 清华: j1 +search 清华大学: j1 +search 科学院: j3 +search 大学: j1 +full 北京: j1 +full 清华: j1 +full 清华大学: j1 +full 科学院: j3 +full 大学: j1 +mix 北京: j1 +mix 清华: (none) +mix 清华大学: j1 +mix 科学院: (none) +mix 大学: (none) +hmm 北京: j1 +hmm 清华: (none) +hmm 清华大学: j1 +hmm 科学院: j3 +hmm 大学: (none) +mixed hello: j1 +mixed 世界: j1 +user 矢量: u1 +dict 矢量: (none) +dict 蓝鲸矢量库: u1 +bad cut_mode: rejected diff --git a/tests/test_fts_tokenizer_ngram.phpt b/tests/test_fts_tokenizer_ngram.phpt new file mode 100644 index 0000000..0d26051 --- /dev/null +++ b/tests/test_fts_tokenizer_ngram.phpt @@ -0,0 +1,131 @@ +--TEST-- +FTS tokenizer "ngram": default bigrams, ngram_min/ngram_max, token_chars, and errors +--SKIPIF-- + +--FILE-- +setTopk(10)->setFts('body', $query, '', $op); + $pks = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $c->queryVector($q)); + sort($pks); + return $pks === [] ? '(none)' : implode(',', $pks); +} + +/** + * @param array $docs + * @return array{0: ZVec, 1: string} collection and its path + */ +function makeCollection(string $name, array $docs, string $extraParams = '', string $tokenizer = 'ngram'): array +{ + $path = __DIR__ . '/../test_dbs/' . $name . '_' . uniqid(); + $schema = new ZVecSchema($name); + $schema->addString('body') + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $c = ZVec::create($path, $schema); + $c->createIndex('body', ZVecIndexParams::forFts( + tokenizer: $tokenizer, + filters: ['lowercase'], + extraParams: $extraParams, + )); + foreach ($docs as $pk => $body) { + $c->insert( + (new ZVecDoc($pk))->setString('body', $body)->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0]) + ); + } + $c->flush(); + return [$c, $path]; +} + +// AND throughout: with OR, "base" is split into ba/as/se and n2 also matches +// through "as", which would hide what the ngram tokenizer actually does. +$and = ZVec::FTS_OPERATOR_AND; + +[$c, $p] = makeCollection('fts_ngram_test', ['n1' => 'database', 'n2' => 'datastore', 'n3' => 'hello']); +try { + echo 'base: ' . ftsPks($c, 'base', $and) . "\n"; + echo 'tab: ' . ftsPks($c, 'tab', $and) . "\n"; + echo 'data: ' . ftsPks($c, 'data', $and) . "\n"; + echo 'xyz: ' . ftsPks($c, 'xyz', $and) . "\n"; + + // The tokenizer config must be persisted, not just cached in memory. + $c->close(); + $c = ZVec::open($p); + echo 'base after reopen: ' . ftsPks($c, 'base', $and) . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +[$c, $p] = makeCollection('fts_ngram_range', ['n1' => 'database', 'n2' => 'datastore'], '{"ngram_min":2,"ngram_max":3}'); +try { + echo '2-3 base: ' . ftsPks($c, 'base', $and) . "\n"; + echo '2-3 sto: ' . ftsPks($c, 'sto', $and) . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +// token_chars restricts what counts as a token character: with ["letter"] the +// digits in "ab12cd" become separators, so the document tokenizes to "ab","cd". +[$c, $p] = makeCollection('fts_ngram_chars', ['n1' => 'ab12cd', 'n2' => 'xy'], '{"token_chars":["letter"]}'); +try { + echo 'chars ab: ' . ftsPks($c, 'ab', $and) . "\n"; + echo 'chars cd: ' . ftsPks($c, 'cd', $and) . "\n"; + echo 'chars b1: ' . ftsPks($c, 'b1', $and) . "\n"; + echo 'chars 12: ' . ftsPks($c, '12', $and) . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($p)); +} + +$errors = [ + 'min>max' => '{"ngram_min":3,"ngram_max":2}', + 'range>1' => '{"ngram_min":1,"ngram_max":3}', + 'min=0' => '{"ngram_min":0}', + 'bad char' => '{"token_chars":["emoji"]}', +]; +foreach ($errors as $label => $extra) { + $path = __DIR__ . '/../test_dbs/fts_ngram_err_' . uniqid(); + try { + $schema = new ZVecSchema('fts_ngram_err'); + $schema->addString('body') + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $c = ZVec::create($path, $schema); + try { + $c->createIndex('body', ZVecIndexParams::forFts( + tokenizer: 'ngram', + filters: ['lowercase'], + extraParams: $extra, + )); + echo "$label: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo "$label: rejected\n"; + } + } finally { + exec('rm -rf ' . escapeshellarg($path)); + } +} +?> +--EXPECT-- +base: n1 +tab: n1 +data: n1,n2 +xyz: (none) +base after reopen: n1 +2-3 base: n1 +2-3 sto: n2 +chars ab: n1 +chars cd: n1 +chars b1: (none) +chars 12: (none) +min>max: rejected +range>1: rejected +min=0: rejected +bad char: rejected