Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- New constants `ZVec::INDEX_TYPE_FTS`, `ZVec::QUERY_PARAM_FTS` (both `11`) plus `FTS_OPERATOR_OR` / `FTS_OPERATOR_AND`.
- FFI: `zvec_index_params_set_fts()`, `zvec_vector_query_set_fts()`, and `IndexType::FTS` handling in the params factory and `to_index_type()`.
- Test: `tests/test_fts.phpt` (index creation, OR/AND semantics, lowercase folding, no-match, input validation).
- Hybrid dense + FTS retrieval through `MultiQuery`, and stemming filters beyond the tested defaults, are not covered yet.
- The `ngram` and `jieba` tokenizers, the `stemmer` filter, and their `extraParams` keys are now covered by `tests/test_fts_tokenizer_ngram.phpt`, `tests/test_fts_tokenizer_jieba.phpt` and `tests/test_fts_filter_stemmer.phpt`. The `forFts()` docblock listed a `stemmer_en` filter and a non-JSON `stemmer_lang=en` example, neither of which upstream accepts; both are now documented correctly and asserted as rejected.
- Hybrid dense + FTS retrieval through `MultiQuery` is not covered yet.

- **DiskANN index type and query params** (#179)
- `ZVecIndexParams::forDiskAnn(metricType, maxDegree, listSize, pqChunkNum, quantizeType)` — the disk-based graph index, previously missing because only the in-memory Vamana variant was exposed. Mirrors the official Go SDK `NewDiskANNIndexParams`.
Expand Down
4 changes: 4 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -522,6 +522,10 @@ $params = ZVecIndexParams::forFts(
// The jieba dictionary ships in zvec_data/jieba_dict next to the library and is
// used automatically. Lookup order: per-field extraParams jieba_dict_dir, then
// ZVEC_JIEBA_DICT_DIR, then ZVec::init(jiebaDictDir:), then the bundled copy.
//
// A bad jieba dictionary path is not a catchable error: cppjieba calls abort(),
// which kills the process with exit code 134. The bundled copy avoids this, and
// ZVec::init(jiebaDictDir:) validates that both dictionary files exist.

// Invert — keyword-based inverted index
$params = ZVecIndexParams::forInvert(
Expand Down
12 changes: 4 additions & 8 deletions src/ZVecIndexParams.php
Original file line number Diff line number Diff line change
Expand Up @@ -199,20 +199,16 @@ public static function forDiskAnn(int $metricType, int $maxDegree = 100, int $li
return new self($handle);
}

/**
* Create Invert index params
*
* @throws ZVecException On FFI error
*/
/**
* Create Full-Text Search index params
*
* Inverted index over a STRING column. Mirrors the official Go SDK
* NewFTSIndexParams(tokenizerName, filters, extraParams).
*
* @param string $tokenizer Tokenizer name ("standard", …)
* @param string[] $filters Token filters, e.g. ["lowercase", "stemmer_en"]
* @param string $extraParams Extra params, e.g. "stemmer_lang=en"
* @param string $tokenizer "standard", "ngram", "jieba" or "whitespace"
* @param string[] $filters Any of "lowercase", "ascii_folding", "stemmer"
* @param string $extraParams JSON object, e.g. '{"stemmer_lang":"english"}',
* '{"ngram_min":2,"ngram_max":3}', '{"cut_mode":"mix"}'
*
* @throws ZVecException On FFI error
*/
Expand Down
140 changes: 140 additions & 0 deletions tests/test_fts_filter_stemmer.phpt
Original file line number Diff line number Diff line change
@@ -0,0 +1,140 @@
--TEST--
FTS filter "stemmer": Snowball stemming, stemmer_lang, and rejected filter names
--SKIPIF--
<?php if (!extension_loaded('ffi')) die('skip FFI extension not available'); ?>
--FILE--
<?php
declare(strict_types=1);
require_once __DIR__ . '/../src/ZVec.php';
// LOG_FATAL: the rejected cases below are *expected* to fail, and upstream logs
// each rejection at ERROR on stderr, which would otherwise land in the test output.
ZVec::init(logType: ZVec::LOG_CONSOLE, logLevel: ZVec::LOG_FATAL);

/**
* @param array<string, string> $docs
* @return array{0: ZVec, 1: string} collection and its path
*/
function makeCollection(string $name, array $docs, array $filters, string $extraParams = ''): array
{
$path = __DIR__ . '/../test_dbs/' . $name . '_' . uniqid();
$schema = new ZVecSchema($name);
$schema->addString('body')
->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP);
$c = ZVec::create($path, $schema);
$c->createIndex('body', ZVecIndexParams::forFts(
tokenizer: 'standard',
filters: $filters,
extraParams: $extraParams,
));
foreach ($docs as $pk => $body) {
$c->insert(
(new ZVecDoc($pk))->setString('body', $body)->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0])
);
}
$c->flush();
return [$c, $path];
}

/** @return string sorted PKs; result order is not part of the contract */
function ftsPks(ZVec $c, string $query): string
{
$q = (new ZVecVectorQuery('body', []))->setTopk(10)->setFts('body', $query);
$pks = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $c->queryVector($q));
sort($pks);
return $pks === [] ? '(none)' : implode(',', $pks);
}

$docs = [
'd1' => 'the quick brown fox jumps',
'd2' => 'foxes are clever animals',
'd3' => 'running runners run daily',
'd4' => 'completely unrelated text',
];

// Without stemming each form is indexed literally, so "fox" and "foxes" are
// different terms. This is the baseline the stemmer below is measured against.
[$c, $p] = makeCollection('fts_stem_base', $docs, ['lowercase']);
try {
echo 'base fox: ' . ftsPks($c, 'fox') . "\n";
echo 'base foxes: ' . ftsPks($c, 'foxes') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
}

// Snowball english is the default language.
[$c, $p] = makeCollection('fts_stem_en', $docs, ['lowercase', 'stemmer']);
try {
echo 'fox: ' . ftsPks($c, 'fox') . "\n";
echo 'foxes: ' . ftsPks($c, 'foxes') . "\n";
echo 'run: ' . ftsPks($c, 'run') . "\n";
echo 'running: ' . ftsPks($c, 'running') . "\n";

$c->close();
$c = ZVec::open($p);
echo 'fox after reopen: ' . ftsPks($c, 'fox') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
}

[$c, $p] = makeCollection('fts_stem_porter', $docs, ['lowercase', 'stemmer'], '{"stemmer_lang":"porter"}');
try {
echo 'porter fox: ' . ftsPks($c, 'fox') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
}

// A different Snowball language, with German umlauts in the documents.
$german = ['d1' => 'Häuser und Katzen', 'd2' => 'Haus'];
[$c, $p] = makeCollection('fts_stem_de', $german, ['lowercase', 'stemmer'], '{"stemmer_lang":"german"}');
try {
echo 'german haus: ' . ftsPks($c, 'haus') . "\n";
echo 'german katze: ' . ftsPks($c, 'katze') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
}

// The two bad examples from the old forFts() docblock, which showed values that
// upstream rejects. Kept here so the docblock fix is backed by a test.
$errors = [
'unknown lang' => [['lowercase', 'stemmer'], '{"stemmer_lang":"klingon"}'],
'stemmer_en' => [['lowercase', 'stemmer_en'], ''],
'not json' => [['lowercase'], 'stemmer_lang=en'],
'bad tokenizer' => [['lowercase'], ''],
];
foreach ($errors as $label => [$filters, $extra]) {
$path = __DIR__ . '/../test_dbs/fts_stem_err_' . uniqid();
try {
$schema = new ZVecSchema('fts_stem_err');
$schema->addString('body')
->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP);
$c = ZVec::create($path, $schema);
try {
$c->createIndex('body', ZVecIndexParams::forFts(
tokenizer: $label === 'bad tokenizer' ? 'bogus' : 'standard',
filters: $filters,
extraParams: $extra,
));
echo "$label: NOT REJECTED\n";
} catch (ZVecException $e) {
echo "$label: rejected\n";
}
} finally {
exec('rm -rf ' . escapeshellarg($path));
}
}
?>
--EXPECT--
base fox: d1
base foxes: d2
fox: d1,d2
foxes: d1,d2
run: d3
running: d3
fox after reopen: d1,d2
porter fox: d1,d2
german haus: d1,d2
german katze: d1
unknown lang: rejected
stemmer_en: rejected
not json: rejected
bad tokenizer: rejected
182 changes: 182 additions & 0 deletions tests/test_fts_tokenizer_jieba.phpt
Original file line number Diff line number Diff line change
@@ -0,0 +1,182 @@
--TEST--
FTS tokenizer "jieba": bundled dictionary, cut modes, user dictionary, and errors
--SKIPIF--
<?php
if (!extension_loaded('ffi')) die('skip FFI extension not available');

$found = false;
foreach (['/../lib/zvec_data/jieba_dict', '/../ffi/build/zvec_data/jieba_dict'] as $d) {
if (is_file(__DIR__ . $d . '/jieba.dict.utf8')) { $found = true; break; }
}
if (!$found) die('skip bundled jieba dictionary not found');
if (getenv('ZVEC_JIEBA_DICT_DIR') !== false) die('skip ZVEC_JIEBA_DICT_DIR overrides the bundled dictionary');
?>
--FILE--
<?php
declare(strict_types=1);
require_once __DIR__ . '/../src/ZVec.php';
// The jieba dictionary is found automatically at init(); no jieba_dict_dir needed.
// LOG_FATAL because the rejected case below is expected to fail and upstream logs
// it at ERROR on stderr.
ZVec::init(logType: ZVec::LOG_CONSOLE, logLevel: ZVec::LOG_FATAL);

/**
* @param array<string, string> $docs
* @return array{0: ZVec, 1: string} collection and its path
*/
function makeCollection(string $name, array $docs, array $filters, string $extraParams = ''): array
{
$path = __DIR__ . '/../test_dbs/' . $name . '_' . uniqid();
$schema = new ZVecSchema($name);
$schema->addString('body')
->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP);
$c = ZVec::create($path, $schema);
$c->createIndex('body', ZVecIndexParams::forFts(
tokenizer: 'jieba',
filters: $filters,
extraParams: $extraParams,
));
foreach ($docs as $pk => $body) {
$c->insert(
(new ZVecDoc($pk))->setString('body', $body)->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0])
);
}
$c->flush();
return [$c, $path];
}

/** @return string sorted PKs; result order is not part of the contract */
function ftsPks(ZVec $c, string $query, ?string $matchString = null): string
{
$q = (new ZVecVectorQuery('body', []))->setTopk(10)
->setFts('body', $query, $matchString ?? '');
$pks = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $c->queryVector($q));
sort($pks);
return $pks === [] ? '(none)' : implode(',', $pks);
}

$docs = [
'j1' => '我来到北京清华大学',
'j2' => '他来到了网易杭研大厦',
'j3' => '小明硕士毕业于中国科学院计算所',
];

// Default cut_mode is "search", which emits both short and long terms.
[$c, $p] = makeCollection('fts_jieba_default', $docs, []);
try {
echo '北京: ' . ftsPks($c, '北京') . "\n";
echo '清华: ' . ftsPks($c, '清华') . "\n";
echo '杭研: ' . ftsPks($c, '杭研') . "\n";
echo '中国科学院: ' . ftsPks($c, '中国科学院') . "\n";
echo '上海: ' . ftsPks($c, '上海') . "\n";
// A phrase query goes through matchString rather than queryString.
echo 'phrase 北京大学: ' . ftsPks($c, '', '北京大学') . "\n";

$c->close();
$c = ZVec::open($p);
echo '北京 after reopen: ' . ftsPks($c, '北京') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
}

// The cut mode really changes the tokens: "mix" favours precise single words and
// drops the compounds that "search" and "full" both emit.
$modes = ['search', 'full', 'mix', 'hmm'];
$terms = ['北京', '清华', '清华大学', '科学院', '大学'];
foreach ($modes as $mode) {
[$c, $p] = makeCollection('fts_jieba_' . $mode, $docs, [], json_encode(['cut_mode' => $mode]));
try {
foreach ($terms as $term) {
echo "$mode $term: " . ftsPks($c, $term) . "\n";
}
} finally {
exec('rm -rf ' . escapeshellarg($p));
}
}

// Mixed Chinese/ASCII text, where lowercase still applies to the ASCII part.
[$c, $p] = makeCollection('fts_jieba_mixed', ['j1' => 'Hello World 你好世界'], ['lowercase']);
try {
echo 'mixed hello: ' . ftsPks($c, 'hello') . "\n";
echo 'mixed 世界: ' . ftsPks($c, '世界') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
}

// A user dictionary adds a term. In "mix" mode, the added whole term then
// matches and the original sub-words stop matching, which is what makes the
// effect observable.
$dictFile = __DIR__ . '/../test_dbs/jieba_user_' . uniqid() . '.dict';
file_put_contents($dictFile, "蓝鲸矢量库 1000 n\n");
[$c, $p] = makeCollection('fts_jieba_user', ['u1' => '蓝鲸矢量库很快'], [], json_encode(['cut_mode' => 'mix']));
try {
echo 'user 矢量: ' . ftsPks($c, '矢量') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
}
[$c, $p] = makeCollection('fts_jieba_user2', ['u1' => '蓝鲸矢量库很快'], [], json_encode(
['cut_mode' => 'mix', 'user_dict_path' => $dictFile],
JSON_UNESCAPED_SLASHES,
));
try {
echo 'dict 矢量: ' . ftsPks($c, '矢量') . "\n";
echo 'dict 蓝鲸矢量库: ' . ftsPks($c, '蓝鲸矢量库') . "\n";
} finally {
exec('rm -rf ' . escapeshellarg($p));
unlink($dictFile);
}

$path = __DIR__ . '/../test_dbs/fts_jieba_err_' . uniqid();
try {
$schema = new ZVecSchema('fts_jieba_err');
$schema->addString('body')
->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP);
$c = ZVec::create($path, $schema);
try {
$c->createIndex('body', ZVecIndexParams::forFts(
tokenizer: 'jieba',
filters: [],
extraParams: '{"cut_mode":"bogus"}',
));
echo "bad cut_mode: NOT REJECTED\n";
} catch (ZVecException $e) {
echo 'bad cut_mode: ' . (str_contains($e->getMessage(), "unknown cut_mode 'bogus'") ? 'rejected' : $e->getMessage()) . "\n";
}
} finally {
exec('rm -rf ' . escapeshellarg($path));
}
?>
--EXPECT--
北京: j1
清华: j1
杭研: j2
中国科学院: j3
上海: (none)
phrase 北京大学: j1
北京 after reopen: j1
search 北京: j1
search 清华: j1
search 清华大学: j1
search 科学院: j3
search 大学: j1
full 北京: j1
full 清华: j1
full 清华大学: j1
full 科学院: j3
full 大学: j1
mix 北京: j1
mix 清华: (none)
mix 清华大学: j1
mix 科学院: (none)
mix 大学: (none)
hmm 北京: j1
hmm 清华: (none)
hmm 清华大学: j1
hmm 科学院: j3
hmm 大学: (none)
mixed hello: j1
mixed 世界: j1
user 矢量: u1
dict 矢量: (none)
dict 蓝鲸矢量库: u1
bad cut_mode: rejected
Loading
Loading