From 4ae380e6a8e77aa10821cdb49985f84485bc7c79 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ha=C5=82as=20Piotr?= Date: Mon, 28 Sep 2026 20:29:57 +0200 Subject: [PATCH 1/3] feat: expose the DiskANN I/O backend (#224) zvec picks an I/O backend for DiskANN disk reads on first use: on Linux it tries io_uring, then libaio, then falls back to synchronous pread(); macOS arm64 always uses pread(). The choice dominates DiskANN throughput -- pread on Linux usually means libaio simply is not installed -- and until now there was no way to observe it from PHP. Three static getters on ZVec, next to getVersion() and following the same pattern the Python SDK uses for its module-level zvec.io_backend_type(): ZVec::getIoBackendType(): int ZVec::getIoBackendTypeName(int $type): string ZVec::getIoBackendDescription(): string plus IO_BACKEND_PREAD / IO_BACKEND_LIBAIO / IO_BACKEND_IO_URING (0/1/2), the same values as the upstream C ABI. The description is where upstream explains how to install an async backend when pread is in use. These are deliberately static rather than per-collection: the value is process-wide, it is not part of GlobalConfig, and upstream probes it lazily, so the getters work without ZVec::init() at all. The test asserts that. Adhering to the #215 singleton rule: the adapter includes only the public zvec/ailego/io/io_backend.h and calls upstream's exported current_io_backend_type() / current_io_backend_description(), which are compiled inside libzvec, so the IOBackend singleton is created only there. The internal io_backend_def.h is not part of the SDK and IOBackend::Instance() is never referenced from this module. The comment in ffi/zvec_ffi.cc records why, so it does not get "fixed" later. $ nm -C ffi/build/libzvec_ffi.so | grep -c 'IOBackend::Instance' 0 $ nm -D --defined-only ffi/build/libzvec_ffi.so | grep io_backend zvec_get_io_backend_description zvec_get_io_backend_type zvec_get_io_backend_type_name No status to check: these return plain values, matching getVersion(), so self::checkStatus() is not involved and FFI::free() is not needed -- the C side owns the memory (a static literal for the name, a thread-local buffer for the description). Tests: 192/192, 0 skipped, 0 failed, 2 expected fail. The new tests/test_io_backend.phpt covers the constants, the name mapping including the "unknown" fallback, that the value is valid before init(), that the description mentions the reported backend, the macOS pread rule, and that the cached value is stable across repeated calls and across init(). tests/test_ffi_load.phpt gained the three symbols (60 -> 63). --- CHANGELOG.md | 6 ++++ README.md | 7 ++++ ffi/zvec_ffi.cc | 27 ++++++++++++++++ ffi/zvec_ffi.h | 12 +++++++ ffi/zvec_ffi_php.h | 3 ++ src/ZVec.php | 66 ++++++++++++++++++++++++++++++++++++++ tests/test_ffi_load.phpt | 5 ++- tests/test_io_backend.phpt | 50 +++++++++++++++++++++++++++++ 8 files changed, 175 insertions(+), 1 deletion(-) create mode 100644 tests/test_io_backend.phpt diff --git a/CHANGELOG.md b/CHANGELOG.md index dad9fd0..df55800 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,12 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **DiskANN I/O backend introspection** (#224) + - `ZVec::getIoBackendType()`, `ZVec::getIoBackendTypeName(int $type)` and `ZVec::getIoBackendDescription()` report the I/O backend zvec picked for DiskANN disk reads. Linux tries `io_uring`, then `libaio`, and falls back to synchronous `pread()`; macOS always uses `pread()`. The value matters a lot for DiskANN throughput and previously could not be observed at all. + - The choice is process-wide and resolved lazily, so the getters work without `ZVec::init()`. New constants `ZVec::IO_BACKEND_PREAD`, `IO_BACKEND_LIBAIO` and `IO_BACKEND_IO_URING` (`0`/`1`/`2`), matching the upstream C ABI. + - FFI: `zvec_get_io_backend_type()`, `zvec_get_io_backend_type_name()` and `zvec_get_io_backend_description()`. The adapter calls upstream's exported `current_io_backend_*()` functions and keeps the #215 singleton rule — the internal `io_backend_def.h` is not part of the SDK and `IOBackend::Instance()` is never referenced here. + - Test: `tests/test_io_backend.phpt` (constant values, name mapping including the `unknown` fallback, value valid before `init()`, description matching the reported name, macOS pread rule, stability across repeated calls and across `init()`). + - **Full-Text Search (FTS) support** (#180) - `ZVecIndexParams::forFts(tokenizer, filters, extraParams)` builds a full-text index over a STRING column; mirrors the official Go SDK `NewFTSIndexParams`. Defaults to the `standard` tokenizer with the `lowercase` filter. - `ZVecVectorQuery::setFts(fieldName, queryString, matchString, defaultOperator)` runs an FTS query. `defaultOperator` accepts `ZVec::FTS_OPERATOR_OR` (default) or `FTS_OPERATOR_AND`, case-insensitively. diff --git a/README.md b/README.md index b6fdb0b..6367cd6 100644 --- a/README.md +++ b/README.md @@ -187,6 +187,9 @@ ZVec::checkVersion(int $major, int $minor, int $patch): bool ZVec::getVersionMajor(): int ZVec::getVersionMinor(): int ZVec::getVersionPatch(): int +ZVec::getIoBackendType(): int // ZVec::IO_BACKEND_PREAD|LIBAIO|IO_URING +ZVec::getIoBackendTypeName(int $type): string // "pread" | "libaio" | "io_uring" | "unknown" +ZVec::getIoBackendDescription(): string // DiskANN I/O backend, with install hints on Linux // Collection lifecycle (static factories) $collection = ZVec::create(string $path, ZVecSchema $schema, bool $readOnly = false, bool $enableMmap = true, int $maxBufferSize = 67108864): self @@ -483,6 +486,9 @@ $params = ZVecIndexParams::forDiskAnn( int $pqChunkNum = 0, int $quantizeType = QUANTIZE_UNDEFINED ): self +// On Linux, check ZVec::getIoBackendType(). IO_BACKEND_PREAD means io_uring and +// libaio are both unavailable, so DiskANN reads are synchronous; install libaio +// (libaio1t64 on Ubuntu 24.04+) for async I/O. // Full-Text Search — inverted index over a STRING column $params = ZVecIndexParams::forFts( @@ -875,6 +881,7 @@ See `tasks/done/` for detailed planning documents. - [x] Array field types (STRING, BOOL, INT32, INT64, UINT32, UINT64, FLOAT, DOUBLE) - [x] Group-by vector query builder (`ZVecGroupByVectorQuery`) - [x] Version API (`getVersion()`, `checkVersion()`) +- [x] DiskANN I/O backend introspection (`getIoBackendType()`, `getIoBackendDescription()`) - [x] `allowedBasePath` security restriction in `init()` - [x] Verbose error details with file/line info - [x] Collection lifecycle options via `getOptions()` diff --git a/ffi/zvec_ffi.cc b/ffi/zvec_ffi.cc index dda34d3..f2de818 100644 --- a/ffi/zvec_ffi.cc +++ b/ffi/zvec_ffi.cc @@ -15,6 +15,7 @@ #include #include #include +#include using namespace zvec; @@ -139,6 +140,32 @@ int zvec_get_version_patch(void) { return kVersionPatch; } +// DiskANN I/O backend. current_io_backend_type() and +// current_io_backend_description() are ordinary functions compiled inside +// libzvec; the IOBackend singleton behind them is created only there, so this +// keeps the #215 singleton rule. Do NOT include the internal +// zvec/ailego/io/io_backend_def.h and do not reference IOBackend::Instance() +// from this module: that header is not part of the SDK, and instantiating the +// singleton here would build a second copy destroyed twice at exit. +int zvec_get_io_backend_type(void) { + return static_cast(zvec::ailego::current_io_backend_type()); +} + +const char* zvec_get_io_backend_type_name(int type) { + switch (type) { + case ZVEC_IO_BACKEND_PREAD: return "pread"; + case ZVEC_IO_BACKEND_LIBAIO: return "libaio"; + case ZVEC_IO_BACKEND_IO_URING: return "io_uring"; + default: return "unknown"; + } +} + +const char* zvec_get_io_backend_description(void) { + static thread_local std::string buf; + buf = zvec::ailego::current_io_backend_description(); + return buf.c_str(); +} + static MetricType to_metric_type(uint32_t v) { switch (v) { case 1: return MetricType::L2; diff --git a/ffi/zvec_ffi.h b/ffi/zvec_ffi.h index d847a41..b0afda4 100644 --- a/ffi/zvec_ffi.h +++ b/ffi/zvec_ffi.h @@ -66,6 +66,18 @@ int zvec_get_version_major(void); int zvec_get_version_minor(void); int zvec_get_version_patch(void); +// DiskANN I/O backend (process-wide, probed lazily; no zvec_init() needed). +// Values match zvec::ailego::IOBackendType and are part of the upstream C ABI. +#define ZVEC_IO_BACKEND_PREAD 0 +#define ZVEC_IO_BACKEND_LIBAIO 1 +#define ZVEC_IO_BACKEND_IO_URING 2 +int zvec_get_io_backend_type(void); +// Returns a static string literal owned by the adapter: "pread", "libaio", +// "io_uring", or "unknown" for any other value. +const char* zvec_get_io_backend_type_name(int type); +// Returns a thread_local buffer, overwritten by the next call in this thread. +const char* zvec_get_io_backend_description(void); + // Global init (call once before any other operation) // log_type: 0=console, 1=file // log_level: 0=DEBUG, 1=INFO, 2=WARN, 3=ERROR, 4=FATAL diff --git a/ffi/zvec_ffi_php.h b/ffi/zvec_ffi_php.h index 0fa414f..f63ecdd 100644 --- a/ffi/zvec_ffi_php.h +++ b/ffi/zvec_ffi_php.h @@ -50,6 +50,9 @@ int zvec_check_version(int major, int minor, int patch); int zvec_get_version_major(void); int zvec_get_version_minor(void); int zvec_get_version_patch(void); +int zvec_get_io_backend_type(void); +const char* zvec_get_io_backend_type_name(int type); +const char* zvec_get_io_backend_description(void); zvec_status_t zvec_init(int log_type, int log_level, const char* log_dir, const char* log_basename, diff --git a/src/ZVec.php b/src/ZVec.php index 6277cdd..0ae8832 100644 --- a/src/ZVec.php +++ b/src/ZVec.php @@ -995,6 +995,35 @@ public function fetch(array|string|bool ...$args): array */ public const QUERY_PARAM_DISKANN = 6; + /** + * DiskANN I/O backend: synchronous pread() reads. + * + * Always used on macOS. On Linux it means neither io_uring nor libaio + * could be loaded, so DiskANN disk reads are synchronous. + * + * Value: 0 + */ + public const IO_BACKEND_PREAD = 0; + + /** + * DiskANN I/O backend: libaio. + * + * Loaded at runtime via dlopen(), so it depends on libaio being installed + * on the host. + * + * Value: 1 + */ + public const IO_BACKEND_LIBAIO = 1; + + /** + * DiskANN I/O backend: io_uring via raw kernel syscalls. + * + * Preferred on Linux; requires kernel 5.1+ and no extra dependency. + * + * Value: 2 + */ + public const IO_BACKEND_IO_URING = 2; + /** * Log destination: Console (stderr). * @@ -1696,6 +1725,43 @@ public static function getVersionPatch(): int return self::ffi()->zvec_get_version_patch(); } + /** + * DiskANN I/O backend in use, one of the ZVec::IO_BACKEND_* constants. + * + * The choice is process-wide and resolved lazily on the first call, so + * this works without ZVec::init(). Linux prefers io_uring, then libaio, + * then falls back to synchronous pread(); macOS always uses pread. + * + * @throws ZVecException On FFI error + */ + public static function getIoBackendType(): int + { + return self::ffi()->zvec_get_io_backend_type(); + } + + /** + * Name of a ZVec::IO_BACKEND_* value, or "unknown" for anything else. + * + * @throws ZVecException On FFI error + */ + public static function getIoBackendTypeName(int $type): string + { + return self::ffi()->zvec_get_io_backend_type_name($type); + } + + /** + * Human-readable description of the active I/O backend. + * + * On Linux the pread variant also explains how to install an async + * backend, so this is the place to look when DiskANN reads seem slow. + * + * @throws ZVecException On FFI error + */ + public static function getIoBackendDescription(): string + { + return self::ffi()->zvec_get_io_backend_description(); + } + /** * Query using a native ZVecVectorQuery object. * diff --git a/tests/test_ffi_load.phpt b/tests/test_ffi_load.phpt index 5873ddd..1042cfa 100644 --- a/tests/test_ffi_load.phpt +++ b/tests/test_ffi_load.phpt @@ -45,6 +45,9 @@ $requiredFunctions = [ 'zvec_get_version_major', 'zvec_get_version_minor', 'zvec_get_version_patch', + 'zvec_get_io_backend_type', + 'zvec_get_io_backend_type_name', + 'zvec_get_io_backend_description', 'zvec_get_last_error_details', 'zvec_clear_error', 'zvec_error_code_to_string', @@ -132,7 +135,7 @@ try { echo "DONE\n"; ?> --EXPECT-- -All 60 FFI symbols resolved successfully +All 63 FFI symbols resolved successfully No FFI::cdef() inline string found in src/ZVec.php Header file zvec_ffi_php.h is used as source of truth Basic create/insert/optimize works diff --git a/tests/test_io_backend.phpt b/tests/test_io_backend.phpt new file mode 100644 index 0000000..0821263 --- /dev/null +++ b/tests/test_io_backend.phpt @@ -0,0 +1,50 @@ +--TEST-- +DiskANN I/O backend: getIoBackendType, getIoBackendTypeName, getIoBackendDescription +--SKIPIF-- + +--FILE-- + +--EXPECT-- +constants: 0,1,2 +names: pread,libaio,io_uring,unknown +type valid: yes +description non-empty: yes +description matches name: yes +platform check: ok +stable: yes +after init: yes +PASS From e0484d449bfa66516a40d9ae291dbe2ededc8732 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ha=C5=82as=20Piotr?= Date: Tue, 29 Sep 2026 08:03:02 +0200 Subject: [PATCH 2/3] feat(init): jiebaDictDir and ftsBruteForceByKeysRatio options (#221) Exposes two upstream GlobalConfig::ConfigData fields that had no PHP equivalent. jiebaDictDir is the folder holding jieba.dict.utf8 and hmm_model.utf8 for the jieba FTS tokenizer; ftsBruteForceByKeysRatio (0.0-1.0, default 0.05) is where an FTS query stops walking posting lists and starts scoring candidates one by one, which pays off when the scalar filter is very selective. Read back with getJiebaDictDir() and getFtsBruteForceByKeysRatio(); with no option given the bundled zvec_data/jieba_dict is still found automatically. null rather than 0.0 as the ratio default, because 0.0 is a real value upstream accepts rather than a "use the default" marker like the other ratios. Both values are validated in PHP before any FFI call, which is the only place the check can live. Upstream GlobalConfig::initialize() sets its initialized flag *before* validating, so a value it rejects still leaves the library marked initialized, with defaults, and every later init() returns OK without applying anything -- a silent misconfiguration. NAN also passes the upstream range test, since NAN < 0 and NAN > 1 are both false. A bad jiebaDictDir is rejected for a second, harder reason: cppjieba calls abort() rather than returning a Status, so a wrong folder kills the process with exit 134 later, when a jieba FTS index is created, with no catchable error. The check therefore requires both dictionary files, not just the directory. (Confirmed: the same abort is reachable through per-field extraParams and through ZVEC_JIEBA_DICT_DIR, which is a separate gap.) The getters use global_config_ptr() rather than GlobalConfig::Instance(), per the #215 singleton rule. The ratio message uses var_export() because interpolating NAN emits a PHP warning; the message is the same shape as the other init() validation errors. Tests: 194/194, 0 skipped, 0 failed, 2 expected fail. The two new tests run in separate processes because upstream config is applied only once per process: - test_init_fts_options.phpt: six rejection cases (1.5, -0.1, NAN, empty dir, missing dir, existing dir without the dictionary files) each asserting isInitialized() is still false afterwards, then a successful init against a *copy* of the bundled dictionary plus an end-to-end jieba FTS query that returns j1, proving a custom folder is the one actually used. - test_init_fts_defaults.phpt: the upstream defaults, 0.05 and the bundled dictionary, with no new options passed. --- CHANGELOG.md | 8 ++ README.md | 13 +++- ffi/zvec_ffi.cc | 35 +++++++++ ffi/zvec_ffi.h | 8 ++ ffi/zvec_ffi_php.h | 4 + src/ZVec.php | 79 +++++++++++++++++++ tests/test_ffi_load.phpt | 6 +- tests/test_init_fts_defaults.phpt | 24 ++++++ tests/test_init_fts_options.phpt | 122 ++++++++++++++++++++++++++++++ 9 files changed, 297 insertions(+), 2 deletions(-) create mode 100644 tests/test_init_fts_defaults.phpt create mode 100644 tests/test_init_fts_options.phpt diff --git a/CHANGELOG.md b/CHANGELOG.md index df55800..aa13e44 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **`ZVec::init()` jieba and FTS tuning options** (#221) + - `?string $jiebaDictDir` sets the folder holding `jieba.dict.utf8` and `hmm_model.utf8` for the jieba FTS tokenizer, and `?float $ftsBruteForceByKeysRatio` (0.0–1.0) the point at which an FTS query stops walking posting lists and scores candidates one by one. Upstream default is 0.05. Both take effect only on the first successful `init()` in a process, like every other option. + - Read back with `ZVec::getJiebaDictDir()` and `ZVec::getFtsBruteForceByKeysRatio()`. With no option given, the jieba dictionary shipped in `zvec_data/jieba_dict` is still picked up automatically. + - `null` rather than `0.0` as the default for the ratio, because 0.0 is a real value upstream accepts. + - Both values are validated in PHP **before** any FFI call. Upstream `GlobalConfig::initialize()` sets its initialized flag before validating, so a rejected value would leave the library marked initialized and make every later `init()` a silent no-op; `NAN` also slips through the upstream range check. A `jiebaDictDir` that lacks the dictionary files is rejected for a worse reason: cppjieba calls `abort()`, killing the process with exit code 134 rather than raising a catchable error. + - FFI: `zvec_config_data_set_fts_brute_force_by_keys_ratio()`, `zvec_config_data_set_jieba_dict_dir()`, `zvec_global_config_get_fts_brute_force_by_keys_ratio()`, `zvec_global_config_get_jieba_dict_dir()`. The getters go through `global_config_ptr()`, keeping the #215 singleton rule. + - Tests: `tests/test_init_fts_options.phpt` (six validation cases that must leave the library uninitialized, plus an end-to-end jieba FTS query using a custom dictionary) and `tests/test_init_fts_defaults.phpt` (upstream defaults). + - **DiskANN I/O backend introspection** (#224) - `ZVec::getIoBackendType()`, `ZVec::getIoBackendTypeName(int $type)` and `ZVec::getIoBackendDescription()` report the I/O backend zvec picked for DiskANN disk reads. Linux tries `io_uring`, then `libaio`, and falls back to synchronous `pread()`; macOS always uses `pread()`. The value matters a lot for DiskANN throughput and previously could not be observed at all. - The choice is process-wide and resolved lazily, so the getters work without `ZVec::init()`. New constants `ZVec::IO_BACKEND_PREAD`, `IO_BACKEND_LIBAIO` and `IO_BACKEND_IO_URING` (`0`/`1`/`2`), matching the upstream C ABI. diff --git a/README.md b/README.md index 6367cd6..4ded94d 100644 --- a/README.md +++ b/README.md @@ -176,9 +176,13 @@ ZVec::init( float $bruteForceByKeysRatio = 0.0, int $memoryLimitMb = 0, ?string $allowedBasePath = null, - bool $verboseErrors = false + bool $verboseErrors = false, + ?string $jiebaDictDir = null, // folder with jieba.dict.utf8 + hmm_model.utf8 + ?float $ftsBruteForceByKeysRatio = null // 0.0-1.0, upstream default 0.05 ): void ZVec::isInitialized(): bool +ZVec::getFtsBruteForceByKeysRatio(): float +ZVec::getJiebaDictDir(): string ZVec::shutdown(): void ZVec::getLastErrorDetails(): array ZVec::clearError(): void @@ -496,6 +500,13 @@ $params = ZVecIndexParams::forFts( string[] $filters = ['lowercase'], string $extraParams = '' ): self +// Tokenizers: "standard", "ngram", "jieba", "whitespace" +// Filters: "lowercase", "ascii_folding", "stemmer" +// extraParams is a JSON object, e.g. '{"stemmer_lang":"english"}', +// '{"ngram_min":2,"ngram_max":3}' or '{"cut_mode":"mix"}' +// The jieba dictionary ships in zvec_data/jieba_dict next to the library and is +// used automatically. Lookup order: per-field extraParams jieba_dict_dir, then +// ZVEC_JIEBA_DICT_DIR, then ZVec::init(jiebaDictDir:), then the bundled copy. // Invert — keyword-based inverted index $params = ZVecIndexParams::forInvert( diff --git a/ffi/zvec_ffi.cc b/ffi/zvec_ffi.cc index f2de818..c37de70 100644 --- a/ffi/zvec_ffi.cc +++ b/ffi/zvec_ffi.cc @@ -373,6 +373,41 @@ void zvec_config_data_set_brute_force_by_keys_ratio(zvec_config_data_t config, f } } +void zvec_config_data_set_fts_brute_force_by_keys_ratio(zvec_config_data_t config, float ratio) { + if (config) { + static_cast(config)->config.fts_brute_force_by_keys_ratio = ratio; + } +} + +void zvec_config_data_set_jieba_dict_dir(zvec_config_data_t config, const char* dir) { + if (config) { + static_cast(config)->config.jieba_dict_dir = dir ? dir : ""; + } +} + +float zvec_global_config_get_fts_brute_force_by_keys_ratio(void) { + auto* gc = global_config_ptr(); + return gc ? gc->fts_brute_force_by_keys_ratio() : 0.0f; +} + +zvec_status_t zvec_global_config_get_jieba_dict_dir(char* buf, size_t buf_size) { + auto* gc = global_config_ptr(); + if (!gc) { + zvec_status_t st = {8, "internal: libzvec GlobalConfig::Instance symbol not found"}; + SET_FFI_ERROR(st); + return st; + } + if (!buf || buf_size == 0) { + zvec_status_t st = {1, "null buffer"}; + SET_FFI_ERROR(st); + return st; + } + std::string dir = gc->jieba_dict_dir(); + strncpy(buf, dir.c_str(), buf_size - 1); + buf[buf_size - 1] = '\0'; + return ok_status(); +} + zvec_status_t zvec_ffi_initialize(zvec_config_data_t config) { auto* gc = global_config_ptr(); if (!gc) { diff --git a/ffi/zvec_ffi.h b/ffi/zvec_ffi.h index b0afda4..2472d9d 100644 --- a/ffi/zvec_ffi.h +++ b/ffi/zvec_ffi.h @@ -106,6 +106,14 @@ void zvec_config_data_set_query_thread_count(zvec_config_data_t config, uint32_t void zvec_config_data_set_optimize_thread_count(zvec_config_data_t config, uint32_t count); void zvec_config_data_set_invert_to_forward_scan_ratio(zvec_config_data_t config, float ratio); void zvec_config_data_set_brute_force_by_keys_ratio(zvec_config_data_t config, float ratio); +void zvec_config_data_set_fts_brute_force_by_keys_ratio(zvec_config_data_t config, float ratio); +// A wrong folder is fatal upstream: cppjieba calls abort() instead of returning +// a Status, so the path is checked here before it can reach libzvec. +void zvec_config_data_set_jieba_dict_dir(zvec_config_data_t config, const char* dir); + +// Read back the effective process-wide values (valid after init()). +float zvec_global_config_get_fts_brute_force_by_keys_ratio(void); +zvec_status_t zvec_global_config_get_jieba_dict_dir(char* buf, size_t buf_size); zvec_status_t zvec_ffi_initialize(zvec_config_data_t config); zvec_status_t zvec_ffi_shutdown(void); diff --git a/ffi/zvec_ffi_php.h b/ffi/zvec_ffi_php.h index f63ecdd..5d4d20d 100644 --- a/ffi/zvec_ffi_php.h +++ b/ffi/zvec_ffi_php.h @@ -77,6 +77,10 @@ void zvec_config_data_set_query_thread_count(zvec_config_data_t config, uint32_t void zvec_config_data_set_optimize_thread_count(zvec_config_data_t config, uint32_t count); void zvec_config_data_set_invert_to_forward_scan_ratio(zvec_config_data_t config, float ratio); void zvec_config_data_set_brute_force_by_keys_ratio(zvec_config_data_t config, float ratio); +void zvec_config_data_set_fts_brute_force_by_keys_ratio(zvec_config_data_t config, float ratio); +void zvec_config_data_set_jieba_dict_dir(zvec_config_data_t config, const char* dir); +float zvec_global_config_get_fts_brute_force_by_keys_ratio(void); +zvec_status_t zvec_global_config_get_jieba_dict_dir(char* buf, size_t buf_size); zvec_status_t zvec_ffi_initialize(zvec_config_data_t config); zvec_status_t zvec_ffi_shutdown(void); diff --git a/src/ZVec.php b/src/ZVec.php index 0ae8832..d55515e 100644 --- a/src/ZVec.php +++ b/src/ZVec.php @@ -1586,12 +1586,41 @@ public static function init( int $memoryLimitMb = 0, ?string $allowedBasePath = null, bool $verboseErrors = false, + ?string $jiebaDictDir = null, + ?float $ftsBruteForceByKeysRatio = null, ): void { self::$verboseErrors = $verboseErrors; if ($allowedBasePath !== null && !is_dir($allowedBasePath)) { throw new ZVecException("Allowed base path does not exist: {$allowedBasePath}"); } + + // Validate before any FFI call. Upstream GlobalConfig::initialize() sets + // its "initialized" flag *before* validating, so a value it rejects still + // leaves the library marked initialized and every later init() silently + // does nothing. NAN also passes the upstream range check. + if ($ftsBruteForceByKeysRatio !== null + && (is_nan($ftsBruteForceByKeysRatio) || $ftsBruteForceByKeysRatio < 0.0 || $ftsBruteForceByKeysRatio > 1.0) + ) { + // var_export(), not interpolation: a NAN would emit a PHP warning here. + throw new ZVecException(sprintf( + 'ftsBruteForceByKeysRatio must be between 0 and 1, got: %s', + var_export($ftsBruteForceByKeysRatio, true) + )); + } + // A bad jieba folder is fatal: cppjieba calls abort() rather than + // returning a Status, killing the process with exit code 134. + if ($jiebaDictDir !== null) { + if ($jiebaDictDir === '' || !is_dir($jiebaDictDir)) { + throw new ZVecException("jiebaDictDir does not exist: {$jiebaDictDir}"); + } + foreach (['jieba.dict.utf8', 'hmm_model.utf8'] as $file) { + if (!is_file($jiebaDictDir . '/' . $file)) { + throw new ZVecException("jiebaDictDir is missing {$file}: {$jiebaDictDir}"); + } + } + } + self::$allowedBasePath = $allowedBasePath; $ffi = self::ffi(); @@ -1618,6 +1647,14 @@ public static function init( if ($memoryLimitMb > 0) { $ffi->zvec_config_data_set_memory_limit($configData, $memoryLimitMb * self::BYTES_PER_MB); } + // null (not 0.0) means "keep the upstream default", because 0.0 is a real + // value here that upstream accepts. + if ($ftsBruteForceByKeysRatio !== null) { + $ffi->zvec_config_data_set_fts_brute_force_by_keys_ratio($configData, $ftsBruteForceByKeysRatio); + } + if ($jiebaDictDir !== null) { + $ffi->zvec_config_data_set_jieba_dict_dir($configData, $jiebaDictDir); + } try { self::checkStatus($ffi->zvec_ffi_initialize($configData)); @@ -1638,6 +1675,48 @@ public static function isInitialized(): bool return self::ffi()->zvec_ffi_is_initialized() !== 0; } + /** + * Effective process-wide FTS candidate-scanning ratio. + * + * Point at which an FTS query stops walking the posting lists and starts + * checking candidates one by one. Separate from + * $bruteForceByKeysRatio because scoring one FTS candidate costs more. + * Upstream default is 0.05. + * + * @throws ZVecException On FFI error + */ + public static function getFtsBruteForceByKeysRatio(): float + { + return self::ffi()->zvec_global_config_get_fts_brute_force_by_keys_ratio(); + } + + /** + * Effective jieba dictionary directory used by the jieba FTS tokenizer. + * + * Lookup order is per-field extraParams jieba_dict_dir, then the + * ZVEC_JIEBA_DICT_DIR environment variable, then this value, then the + * dictionary shipped next to the library. + * + * @throws ZVecException On FFI error + */ + public static function getJiebaDictDir(): string + { + $ffi = self::ffi(); + $bufSize = self::PATH_BUFFER_SIZE; + while (true) { + $buf = $ffi->new("char[$bufSize]"); + self::checkStatus($ffi->zvec_global_config_get_jieba_dict_dir($buf, $bufSize)); + $str = FFI::string($buf); + if (strlen($str) < $bufSize - 1) { + return $str; + } + $bufSize *= 2; + if ($bufSize > self::MAX_STRING_BUFFER_SIZE) { + throw new ZVecException('jiebaDictDir string exceeds maximum buffer size of 1 MB'); + } + } + } + /** * Shut down the zvec library and release global resources. * diff --git a/tests/test_ffi_load.phpt b/tests/test_ffi_load.phpt index 1042cfa..39c4b0f 100644 --- a/tests/test_ffi_load.phpt +++ b/tests/test_ffi_load.phpt @@ -48,6 +48,10 @@ $requiredFunctions = [ 'zvec_get_io_backend_type', 'zvec_get_io_backend_type_name', 'zvec_get_io_backend_description', + 'zvec_config_data_set_fts_brute_force_by_keys_ratio', + 'zvec_config_data_set_jieba_dict_dir', + 'zvec_global_config_get_fts_brute_force_by_keys_ratio', + 'zvec_global_config_get_jieba_dict_dir', 'zvec_get_last_error_details', 'zvec_clear_error', 'zvec_error_code_to_string', @@ -135,7 +139,7 @@ try { echo "DONE\n"; ?> --EXPECT-- -All 63 FFI symbols resolved successfully +All 67 FFI symbols resolved successfully No FFI::cdef() inline string found in src/ZVec.php Header file zvec_ffi_php.h is used as source of truth Basic create/insert/optimize works diff --git a/tests/test_init_fts_defaults.phpt b/tests/test_init_fts_defaults.phpt new file mode 100644 index 0000000..ad8a736 --- /dev/null +++ b/tests/test_init_fts_defaults.phpt @@ -0,0 +1,24 @@ +--TEST-- +ZVec::init(): upstream defaults for jiebaDictDir and ftsBruteForceByKeysRatio +--SKIPIF-- + +--FILE-- + +--EXPECT-- +ratio: 0.05 +jieba dict: bundled +is initialized: yes diff --git a/tests/test_init_fts_options.phpt b/tests/test_init_fts_options.phpt new file mode 100644 index 0000000..ec58aa0 --- /dev/null +++ b/tests/test_init_fts_options.phpt @@ -0,0 +1,122 @@ +--TEST-- +ZVec::init(): jiebaDictDir and ftsBruteForceByKeysRatio options, with getters +--SKIPIF-- + +--FILE-- + 1.5, + 'negative' => -0.1, + 'NaN' => NAN, +] as $label => $ratio) { + try { + ZVec::init(ftsBruteForceByKeysRatio: $ratio); + echo "$label: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo "$label: rejected\n"; + } +} + +foreach ([ + 'empty dir' => '', + 'missing dir' => '/nonexistent/zvec/jieba', +] as $label => $dir) { + try { + ZVec::init(jiebaDictDir: $dir); + echo "$label: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo "$label: rejected\n"; + } +} + +// A folder that exists but lacks the dictionary files would be fatal upstream: +// cppjieba calls abort() instead of returning a Status, killing the process. +$emptyDir = __DIR__ . '/../test_dbs/jieba_empty_' . uniqid(); +mkdir($emptyDir, 0777, true); +try { + ZVec::init(jiebaDictDir: $emptyDir); + echo "incomplete dir: NOT REJECTED\n"; +} catch (ZVecException $e) { + echo "incomplete dir: rejected\n"; +} finally { + exec('rm -rf ' . escapeshellarg($emptyDir)); +} + +echo 'pre-init: ' . (ZVec::isInitialized() ? '1' : '0') . "\n"; + +// Now init for real with a *copy* of the bundled dictionary, so the test proves +// a custom folder is honoured rather than the bundled default. +$customDir = __DIR__ . '/../test_dbs/jieba_custom_' . uniqid(); +mkdir($customDir, 0777, true); +$source = is_file(__DIR__ . '/../lib/zvec_data/jieba_dict/jieba.dict.utf8') + ? __DIR__ . '/../lib/zvec_data/jieba_dict' + : __DIR__ . '/../ffi/build/zvec_data/jieba_dict'; +foreach (['jieba.dict.utf8', 'hmm_model.utf8'] as $file) { + copy("{$source}/{$file}", "{$customDir}/{$file}"); +} + +$path = __DIR__ . '/../test_dbs/init_fts_' . uniqid(); +try { + ZVec::init( + logType: ZVec::LOG_CONSOLE, + logLevel: ZVec::LOG_WARN, + jiebaDictDir: $customDir, + ftsBruteForceByKeysRatio: 0.2, + ); + + // The value is stored as a C float, so compare as a string, not with ===. + echo 'ratio: ' . sprintf('%.2f', ZVec::getFtsBruteForceByKeysRatio()) . "\n"; + echo 'dir matches: ' . (ZVec::getJiebaDictDir() === $customDir ? 'yes' : 'no') . "\n"; + + // End to end: an FTS index using the jieba tokenizer without a per-field + // jieba_dict_dir, so the global value is what actually resolves it. + $schema = new ZVecSchema('jieba_init_test'); + $schema->addString('body') + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + + $collection = ZVec::create($path, $schema); + $collection->createIndex('body', ZVecIndexParams::forFts(tokenizer: 'jieba', filters: [])); + $collection->insert( + (new ZVecDoc('j1'))->setString('body', '我来到北京清华大学')->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0]) + ); + $collection->insert( + (new ZVecDoc('j2'))->setString('body', '他来到了网易杭研大厦')->setVectorFp32('vec', [0.0, 1.0, 0.0, 0.0]) + ); + $collection->flush(); + + $query = (new ZVecVectorQuery('body', []))->setTopk(10)->setFts('body', '北京'); + $hits = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $collection->queryVector($query)); + echo 'fts hit: ' . implode(',', $hits) . "\n"; +} finally { + exec('rm -rf ' . escapeshellarg($path)); + exec('rm -rf ' . escapeshellarg($customDir)); +} +?> +--EXPECT-- +too high: rejected +negative: rejected +NaN: rejected +empty dir: rejected +missing dir: rejected +incomplete dir: rejected +pre-init: 0 +ratio: 0.20 +dir matches: yes +fts hit: j1 From 67240587560cbd5cb112a9a2e3515b3283bfb748 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ha=C5=82as=20Piotr?= Date: Tue, 29 Sep 2026 08:09:05 +0200 Subject: [PATCH 3/3] feat(schema): declare an FTS or invert index in addString() (#223) Declaring a full-text index on a STRING field took two steps: create the collection, then call createIndex(). Upstream puts the index params on the field itself, so the index can exist from the first insert: $schema->addString('body', indexParams: ZVecIndexParams::forFts()); $schema->addString('tag', indexParams: ZVecIndexParams::forInvert()); This is what the Python SDK does with FieldSchema(index_param=...), and what upstream's own hybrid tests do. The new argument is last and optional, so existing calls are unaffected. Two validation cases the older signature could not express. Passing both $withInvertIndex and $indexParams is ambiguous and now throws. And unlike the other add*() methods, a duplicate field name is reported here rather than silently dropped: upstream returns AlreadyExists from add_field(), and zvec_schema_add_field_* discards that Status, so addString('body') twice currently loses a field without a word. This function propagates it because it has to build index params anyway. A mismatched index type is deliberately *not* rejected here. Upstream reports it at ZVec::create() time with a message naming the field ("scalar field[body] does not support vector index params, but got index_type ..."), which is clearer than anything the adapter could produce, so there is one source of truth. Upstream's FieldSchema constructor clones the index params, so the ZVecIndexParams object keeps owning its handle and is freed by its own destructor as usual; nothing new to release on the PHP side. FFI: zvec_schema_add_field_string_with_index(), declared next to zvec_collection_create_index() in both headers because the zvec_index_params_t typedef comes after the schema block. Tests: 195/195, 0 skipped, 0 failed, 2 expected fail. The new tests/test_schema_fts_index.phpt queries for a term without any createIndex() call, checks getIndexType() is 11 before and after a reopen, covers invert via params (10), both validation errors, and the pre-existing argument combinations. test_ffi_load.phpt gained the symbol (67 -> 68). Note: tests/test_diskann_index.phpt was seen failing once with only doc2 returned instead of doc1,doc2,doc3, and passed on five subsequent single runs and four subsequent full-suite runs. Not reproduced and not caused by this change; a flaky DiskANN query result on a 3-document collection, worth a separate look if it shows up again. --- CHANGELOG.md | 11 +++ README.md | 7 +- ffi/zvec_ffi.cc | 14 ++++ ffi/zvec_ffi.h | 9 +++ ffi/zvec_ffi_php.h | 1 + src/ZVecSchema.php | 39 ++++++++-- tests/test_ffi_load.phpt | 3 +- tests/test_schema_fts_index.phpt | 122 +++++++++++++++++++++++++++++++ 8 files changed, 199 insertions(+), 7 deletions(-) create mode 100644 tests/test_schema_fts_index.phpt diff --git a/CHANGELOG.md b/CHANGELOG.md index aa13e44..4bce9e5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **Scalar index declared in `ZVecSchema::addString()`** (#223) + - `addString()` takes an optional `?ZVecIndexParams $indexParams`, so a full-text or invert index can be part of the schema instead of a separate `createIndex()` call after the collection exists: + ```php + $schema->addString('body', indexParams: ZVecIndexParams::forFts()); + ``` + The index then exists from the first insert. Mirrors the Python SDK's `FieldSchema(index_param=...)` and how upstream's own hybrid tests build collections. + - Passing both `$withInvertIndex` and `$indexParams` throws, and unlike the other `add*()` methods a duplicate field name is now reported instead of silently dropped (upstream returns `AlreadyExists`; the older `add*()` functions discard that `Status`). + - A mismatched index type is left to upstream, which reports it at `ZVec::create()` time naming the field, so there is one source of truth rather than two. + - FFI: `zvec_schema_add_field_string_with_index()`. The index params are cloned by upstream's `FieldSchema` constructor, so the `ZVecIndexParams` object may be freed as usual. + - Test: `tests/test_schema_fts_index.phpt` (query without `createIndex()`, introspection before and after reopen, invert via params, the two validation errors, and backward compatibility of the existing arguments). + - **`ZVec::init()` jieba and FTS tuning options** (#221) - `?string $jiebaDictDir` sets the folder holding `jieba.dict.utf8` and `hmm_model.utf8` for the jieba FTS tokenizer, and `?float $ftsBruteForceByKeysRatio` (0.0–1.0) the point at which an FTS query stops walking posting lists and scores candidates one by one. Upstream default is 0.05. Both take effect only on the first successful `init()` in a process, like every other option. - Read back with `ZVec::getJiebaDictDir()` and `ZVec::getFtsBruteForceByKeysRatio()`. With no option given, the jieba dictionary shipped in `zvec_data/jieba_dict` is still picked up automatically. diff --git a/README.md b/README.md index 4ded94d..35f12d8 100644 --- a/README.md +++ b/README.md @@ -274,7 +274,7 @@ $schema->setMaxDocCountPerSegment(int $count): self // Scalar fields $schema->addInt64(string $name, bool $nullable = false, bool $withInvertIndex = false): self -$schema->addString(string $name, bool $nullable = false, bool $withInvertIndex = false): self +$schema->addString(string $name, bool $nullable = false, bool $withInvertIndex = false, ?ZVecIndexParams $indexParams = null): self $schema->addFloat(string $name, bool $nullable = true): self $schema->addDouble(string $name, bool $nullable = true): self $schema->addBool(string $name, bool $nullable = false, bool $withInvertIndex = false): self @@ -500,6 +500,11 @@ $params = ZVecIndexParams::forFts( string[] $filters = ['lowercase'], string $extraParams = '' ): self +// The same index can be declared up front, so it exists from the first insert +// instead of via a later createIndex() call: +// $schema->addString('body', indexParams: ZVecIndexParams::forFts()); +// $schema->addString('tag', indexParams: ZVecIndexParams::forInvert()); +// // Tokenizers: "standard", "ngram", "jieba", "whitespace" // Filters: "lowercase", "ascii_folding", "stemmer" // extraParams is a JSON object, e.g. '{"stemmer_lang":"english"}', diff --git a/ffi/zvec_ffi.cc b/ffi/zvec_ffi.cc index c37de70..9453424 100644 --- a/ffi/zvec_ffi.cc +++ b/ffi/zvec_ffi.cc @@ -1272,6 +1272,20 @@ zvec_status_t zvec_collection_create_index(zvec_collection_t coll, const char* f return MAKE_STATUS(c->create_index(field_name, index_params, opts)); } +zvec_status_t zvec_schema_add_field_string_with_index(zvec_schema_t schema, const char* name, int nullable, zvec_index_params_t params) { + if (!schema || !name || !params) { + return MAKE_STATUS(Status(StatusCode::INVALID_ARGUMENT, "null schema, name or index params")); + } + auto* s = static_cast(schema); + auto index_params = static_cast(params)->build(); + if (!index_params) { + return MAKE_STATUS(Status(StatusCode::INVALID_ARGUMENT, "Invalid or unsupported index type")); + } + // FieldSchema's constructor clones index_params, so the caller's ZVecIndexParams + // stays valid and frees its own handle whenever it goes out of scope. + return MAKE_STATUS(s->add_field(std::make_shared(name, DataType::STRING, (bool)nullable, index_params))); +} + // --- Doc --- zvec_doc_t zvec_doc_create(const char* pk) { diff --git a/ffi/zvec_ffi.h b/ffi/zvec_ffi.h index 2472d9d..312868c 100644 --- a/ffi/zvec_ffi.h +++ b/ffi/zvec_ffi.h @@ -215,6 +215,15 @@ void zvec_index_params_set_quantizer_enable_rotate(zvec_index_params_t params, i void zvec_index_params_set_metric_type(zvec_index_params_t params, int metric_type); zvec_status_t zvec_collection_create_index(zvec_collection_t coll, const char* field_name, zvec_index_params_t params, uint32_t concurrency); +// Adds a STRING field carrying an explicit scalar index (FTS or INVERT), so the +// index exists from the first insert instead of a later create_index() call. +// params is consumed by this call and may be freed right afterwards. Declared +// here rather than next to zvec_schema_add_field_string because the +// zvec_index_params_t typedef comes later in this header. +// A vector index type is not rejected here: upstream reports it at create time +// with a clearer message, and one source of truth is preferable. +zvec_status_t zvec_schema_add_field_string_with_index(zvec_schema_t schema, const char* name, int nullable, zvec_index_params_t params); + // Doc zvec_doc_t zvec_doc_create(const char* pk); void zvec_doc_free(zvec_doc_t doc); diff --git a/ffi/zvec_ffi_php.h b/ffi/zvec_ffi_php.h index 5d4d20d..15963dd 100644 --- a/ffi/zvec_ffi_php.h +++ b/ffi/zvec_ffi_php.h @@ -170,6 +170,7 @@ void zvec_index_params_set_quantize_type(zvec_index_params_t params, int quantiz void zvec_index_params_set_quantizer_enable_rotate(zvec_index_params_t params, int enable_rotate); void zvec_index_params_set_metric_type(zvec_index_params_t params, int metric_type); zvec_status_t zvec_collection_create_index(zvec_collection_t coll, const char* field_name, zvec_index_params_t params, uint32_t concurrency); +zvec_status_t zvec_schema_add_field_string_with_index(zvec_schema_t schema, const char* name, int nullable, zvec_index_params_t params); zvec_doc_t zvec_doc_create(const char* pk); void zvec_doc_free(zvec_doc_t doc); diff --git a/src/ZVecSchema.php b/src/ZVecSchema.php index 3c00200..d2fecc8 100644 --- a/src/ZVecSchema.php +++ b/src/ZVecSchema.php @@ -69,11 +69,40 @@ public function addInt64(string $name, bool $nullable = false, bool $withInvertI } /** - * @throws ZVecException On FFI error - */ - public function addString(string $name, bool $nullable = false, bool $withInvertIndex = false): self - { - self::ffi()->zvec_schema_add_field_string($this->handle, $name, $nullable ? 1 : 0, $withInvertIndex ? 1 : 0); + * Adds a STRING field, optionally carrying a scalar index so it exists from + * the first insert instead of a later createIndex() call. + * + * Pass $indexParams for a full-text or invert index: + * + * $schema->addString('body', indexParams: ZVecIndexParams::forFts()); + * + * That is an alternative to creating the collection and then calling + * createIndex('body', ...). Upstream validates a mismatched index type at + * ZVec::create() time, so passing vector params fails there with a message + * naming the field. + * + * @throws ZVecException On FFI error, duplicate field, or when both + * $withInvertIndex and $indexParams are given + */ + public function addString( + string $name, + bool $nullable = false, + bool $withInvertIndex = false, + ?ZVecIndexParams $indexParams = null, + ): self { + if ($indexParams === null) { + self::ffi()->zvec_schema_add_field_string($this->handle, $name, $nullable ? 1 : 0, $withInvertIndex ? 1 : 0); + return $this; + } + if ($withInvertIndex) { + throw new ZVecException('Use either $withInvertIndex or $indexParams, not both'); + } + ZVec::checkStatus(self::ffi()->zvec_schema_add_field_string_with_index( + $this->handle, + $name, + $nullable ? 1 : 0, + $indexParams->getHandle(), + )); return $this; } diff --git a/tests/test_ffi_load.phpt b/tests/test_ffi_load.phpt index 39c4b0f..5abf265 100644 --- a/tests/test_ffi_load.phpt +++ b/tests/test_ffi_load.phpt @@ -14,6 +14,7 @@ $requiredFunctions = [ 'zvec_init', 'zvec_schema_create', 'zvec_schema_free', + 'zvec_schema_add_field_string_with_index', 'zvec_collection_create', 'zvec_collection_open', 'zvec_collection_free', @@ -139,7 +140,7 @@ try { echo "DONE\n"; ?> --EXPECT-- -All 67 FFI symbols resolved successfully +All 68 FFI symbols resolved successfully No FFI::cdef() inline string found in src/ZVec.php Header file zvec_ffi_php.h is used as source of truth Basic create/insert/optimize works diff --git a/tests/test_schema_fts_index.phpt b/tests/test_schema_fts_index.phpt new file mode 100644 index 0000000..d064cf4 --- /dev/null +++ b/tests/test_schema_fts_index.phpt @@ -0,0 +1,122 @@ +--TEST-- +Schema: declare an FTS or invert index on a STRING field at create time +--SKIPIF-- + +--FILE-- +setTopk(10)->setFts('body', $term); + $pks = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $collection->queryVector($query)); + sort($pks); + return $pks; +} + +$path = __DIR__ . '/../test_dbs/schema_fts_' . uniqid(); +$invertPath = __DIR__ . '/../test_dbs/schema_invert_' . uniqid(); +$hnswPath = __DIR__ . '/../test_dbs/schema_hnsw_' . uniqid(); + +try { + $schema = new ZVecSchema('schema_fts_test'); + $schema->addString('body', indexParams: ZVecIndexParams::forFts()) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + + $collection = ZVec::create($path, $schema); + + foreach ([ + ['d1', 'the quick brown fox jumps'], + ['d2', 'a fast brown rabbit'], + ['d3', 'completely unrelated text about databases'], + ] as [$pk, $body]) { + $collection->insert( + (new ZVecDoc($pk))->setString('body', $body)->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0]) + ); + } + $collection->flush(); + + // The index exists from the first insert: no createIndex() call anywhere. + echo 'fts in schema: ' . implode(',', ftsPks($collection, 'fox')) . "\n"; + echo 'index type: ' . $collection->getFieldSchema('body')->getIndexType() . "\n"; + + $collection->close(); + $collection = ZVec::open($path); + echo 'after reopen: ' . implode(',', ftsPks($collection, 'fox')) . "\n"; + echo 'index type after reopen: ' . $collection->getFieldSchema('body')->getIndexType() . "\n"; + + // Invert index declared the same way. + $invertSchema = new ZVecSchema('schema_invert_test'); + $invertSchema->addString('tag', indexParams: ZVecIndexParams::forInvert()) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $invertCollection = ZVec::create($invertPath, $invertSchema); + echo 'invert type: ' . $invertCollection->getFieldSchema('tag')->getIndexType() . "\n"; + + // Requesting both an invert flag and explicit params is ambiguous. + try { + (new ZVecSchema('schema_conflict'))->addString( + 'a', + withInvertIndex: true, + indexParams: ZVecIndexParams::forFts(), + ); + echo "both flags: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo 'both flags: ' . $e->getMessage() . "\n"; + } + + // A duplicate field is reported here, unlike the other add*() methods which + // drop the Status from add_field(). + try { + $duplicate = new ZVecSchema('schema_duplicate'); + $duplicate->addString('body'); + $duplicate->addString('body', indexParams: ZVecIndexParams::forFts()); + echo "duplicate: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo 'duplicate: ' . (str_contains($e->getMessage(), 'already exists') ? 'rejected' : $e->getMessage()) . "\n"; + } + + // Vector index params on a scalar field are caught by upstream at create + // time, which names the field and the type. + $hnswSchema = new ZVecSchema('schema_hnsw_string'); + $hnswSchema->addString('body', indexParams: ZVecIndexParams::forHnsw(ZVecSchema::METRIC_IP)) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + try { + ZVec::create($hnswPath, $hnswSchema); + echo "hnsw on string: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo 'hnsw on string: ' . (str_contains($e->getMessage(), 'does not support vector index params') + ? 'rejected' : $e->getMessage()) . "\n"; + } + + // Backward compatibility of the pre-existing argument combinations. + $bcSchema = new ZVecSchema('schema_bc'); + $bcSchema->addString('x') + ->addString('x2', withInvertIndex: true) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $bcPath = __DIR__ . '/../test_dbs/schema_bc_' . uniqid(); + $bcCollection = ZVec::create($bcPath, $bcSchema); + echo 'bc: ' . ($bcCollection->getFieldSchema('x2')->hasInvertIndex() ? 'ok' : 'FAILED') . "\n"; + $bcCollection->close(); + exec('rm -rf ' . escapeshellarg($bcPath)); + + echo "PASS\n"; +} finally { + exec('rm -rf ' . escapeshellarg($path)); + exec('rm -rf ' . escapeshellarg($invertPath)); + exec('rm -rf ' . escapeshellarg($hnswPath)); +} +?> +--EXPECT-- +fts in schema: d1 +index type: 11 +after reopen: d1 +index type after reopen: 11 +invert type: 10 +both flags: Use either $withInvertIndex or $indexParams, not both +duplicate: rejected +hnsw on string: rejected +bc: ok +PASS