diff --git a/CHANGELOG.md b/CHANGELOG.md index aacff9e..9ce80ac 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Added +- **Scalar index declared in `ZVecSchema::addString()`** (#223) + - `addString()` takes an optional `?ZVecIndexParams $indexParams`, so a full-text or invert index can be part of the schema instead of a separate `createIndex()` call after the collection exists: + ```php + $schema->addString('body', indexParams: ZVecIndexParams::forFts()); + ``` + The index then exists from the first insert. Mirrors the Python SDK's `FieldSchema(index_param=...)` and how upstream's own hybrid tests build collections. + - Passing both `$withInvertIndex` and `$indexParams` throws, and unlike the other `add*()` methods a duplicate field name is now reported instead of silently dropped (upstream returns `AlreadyExists`; the older `add*()` functions discard that `Status`). + - A mismatched index type is left to upstream, which reports it at `ZVec::create()` time naming the field, so there is one source of truth rather than two. + - FFI: `zvec_schema_add_field_string_with_index()`. The index params are cloned by upstream's `FieldSchema` constructor, so the `ZVecIndexParams` object may be freed as usual. + - Test: `tests/test_schema_fts_index.phpt` (query without `createIndex()`, introspection before and after reopen, invert via params, the two validation errors, and backward compatibility of the existing arguments). - **`ZVec::init()` jieba and FTS tuning options** (#221) - `?string $jiebaDictDir` sets the folder holding `jieba.dict.utf8` and `hmm_model.utf8` for the jieba FTS tokenizer, and `?float $ftsBruteForceByKeysRatio` (0.0–1.0) the point at which an FTS query stops walking posting lists and scores candidates one by one. Upstream default is 0.05. Both take effect only on the first successful `init()` in a process, like every other option. - Read back with `ZVec::getJiebaDictDir()` and `ZVec::getFtsBruteForceByKeysRatio()`. With no option given, the jieba dictionary shipped in `zvec_data/jieba_dict` is still picked up automatically. @@ -25,7 +35,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - FFI: `zvec_index_params_set_ivf_rabitq()`, `zvec_vector_query_set_ivf_rabitq_nprobe()`, the index type in `to_index_type()` / `from_index_type()`, an `IVF_RABITQ` case in `ensure_query_params_for_field()` and the group-by switch, and a `case 7` in the legacy `query()` path's `validate_query_param_type()` and `apply_query_params()`. The legacy path reuses the existing `ivfNprobe` argument, having no separate one. - `scale_factor` is deliberately not exposed: the upstream engine copies only `nprobe` for this index type, so exposing it would be misleading. - Tests: `tests/test_ivf_rabitq_params.phpt` (constants, validation, query-param plumbing) and `tests/test_ivf_rabitq_index.phpt` (real index, both query paths, radius-only through `ensure_query_params_for_field()`, a params-type mismatch, and the dimension rule). - - **DiskANN I/O backend introspection** (#224) - `ZVec::getIoBackendType()`, `ZVec::getIoBackendTypeName(int $type)` and `ZVec::getIoBackendDescription()` report the I/O backend zvec picked for DiskANN disk reads. Linux tries `io_uring`, then `libaio`, and falls back to synchronous `pread()`; macOS always uses `pread()`. The value matters a lot for DiskANN throughput and previously could not be observed at all. - The choice is process-wide and resolved lazily, so the getters work without `ZVec::init()`. New constants `ZVec::IO_BACKEND_PREAD`, `IO_BACKEND_LIBAIO` and `IO_BACKEND_IO_URING` (`0`/`1`/`2`), matching the upstream C ABI. diff --git a/README.md b/README.md index a0b49e0..a2be5ac 100644 --- a/README.md +++ b/README.md @@ -274,7 +274,7 @@ $schema->setMaxDocCountPerSegment(int $count): self // Scalar fields $schema->addInt64(string $name, bool $nullable = false, bool $withInvertIndex = false): self -$schema->addString(string $name, bool $nullable = false, bool $withInvertIndex = false): self +$schema->addString(string $name, bool $nullable = false, bool $withInvertIndex = false, ?ZVecIndexParams $indexParams = null): self $schema->addFloat(string $name, bool $nullable = true): self $schema->addDouble(string $name, bool $nullable = true): self $schema->addBool(string $name, bool $nullable = false, bool $withInvertIndex = false): self @@ -510,6 +510,11 @@ $params = ZVecIndexParams::forFts( string[] $filters = ['lowercase'], string $extraParams = '' ): self +// The same index can be declared up front, so it exists from the first insert +// instead of via a later createIndex() call: +// $schema->addString('body', indexParams: ZVecIndexParams::forFts()); +// $schema->addString('tag', indexParams: ZVecIndexParams::forInvert()); +// // Tokenizers: "standard", "ngram", "jieba", "whitespace" // Filters: "lowercase", "ascii_folding", "stemmer" // extraParams is a JSON object, e.g. '{"stemmer_lang":"english"}', diff --git a/ffi/zvec_ffi.cc b/ffi/zvec_ffi.cc index c516903..141535d 100644 --- a/ffi/zvec_ffi.cc +++ b/ffi/zvec_ffi.cc @@ -1289,6 +1289,20 @@ zvec_status_t zvec_collection_create_index(zvec_collection_t coll, const char* f return MAKE_STATUS(c->create_index(field_name, index_params, opts)); } +zvec_status_t zvec_schema_add_field_string_with_index(zvec_schema_t schema, const char* name, int nullable, zvec_index_params_t params) { + if (!schema || !name || !params) { + return MAKE_STATUS(Status(StatusCode::INVALID_ARGUMENT, "null schema, name or index params")); + } + auto* s = static_cast(schema); + auto index_params = static_cast(params)->build(); + if (!index_params) { + return MAKE_STATUS(Status(StatusCode::INVALID_ARGUMENT, "Invalid or unsupported index type")); + } + // FieldSchema's constructor clones index_params, so the caller's ZVecIndexParams + // stays valid and frees its own handle whenever it goes out of scope. + return MAKE_STATUS(s->add_field(std::make_shared(name, DataType::STRING, (bool)nullable, index_params))); +} + // --- Doc --- zvec_doc_t zvec_doc_create(const char* pk) { diff --git a/ffi/zvec_ffi.h b/ffi/zvec_ffi.h index 2116dc3..d41b888 100644 --- a/ffi/zvec_ffi.h +++ b/ffi/zvec_ffi.h @@ -216,6 +216,15 @@ void zvec_index_params_set_quantizer_enable_rotate(zvec_index_params_t params, i void zvec_index_params_set_metric_type(zvec_index_params_t params, int metric_type); zvec_status_t zvec_collection_create_index(zvec_collection_t coll, const char* field_name, zvec_index_params_t params, uint32_t concurrency); +// Adds a STRING field carrying an explicit scalar index (FTS or INVERT), so the +// index exists from the first insert instead of a later create_index() call. +// params is consumed by this call and may be freed right afterwards. Declared +// here rather than next to zvec_schema_add_field_string because the +// zvec_index_params_t typedef comes later in this header. +// A vector index type is not rejected here: upstream reports it at create time +// with a clearer message, and one source of truth is preferable. +zvec_status_t zvec_schema_add_field_string_with_index(zvec_schema_t schema, const char* name, int nullable, zvec_index_params_t params); + // Doc zvec_doc_t zvec_doc_create(const char* pk); void zvec_doc_free(zvec_doc_t doc); diff --git a/ffi/zvec_ffi_php.h b/ffi/zvec_ffi_php.h index 7a1d6e4..c2dfae2 100644 --- a/ffi/zvec_ffi_php.h +++ b/ffi/zvec_ffi_php.h @@ -171,6 +171,7 @@ void zvec_index_params_set_quantize_type(zvec_index_params_t params, int quantiz void zvec_index_params_set_quantizer_enable_rotate(zvec_index_params_t params, int enable_rotate); void zvec_index_params_set_metric_type(zvec_index_params_t params, int metric_type); zvec_status_t zvec_collection_create_index(zvec_collection_t coll, const char* field_name, zvec_index_params_t params, uint32_t concurrency); +zvec_status_t zvec_schema_add_field_string_with_index(zvec_schema_t schema, const char* name, int nullable, zvec_index_params_t params); zvec_doc_t zvec_doc_create(const char* pk); void zvec_doc_free(zvec_doc_t doc); diff --git a/src/ZVecSchema.php b/src/ZVecSchema.php index 3c00200..d2fecc8 100644 --- a/src/ZVecSchema.php +++ b/src/ZVecSchema.php @@ -69,11 +69,40 @@ public function addInt64(string $name, bool $nullable = false, bool $withInvertI } /** - * @throws ZVecException On FFI error - */ - public function addString(string $name, bool $nullable = false, bool $withInvertIndex = false): self - { - self::ffi()->zvec_schema_add_field_string($this->handle, $name, $nullable ? 1 : 0, $withInvertIndex ? 1 : 0); + * Adds a STRING field, optionally carrying a scalar index so it exists from + * the first insert instead of a later createIndex() call. + * + * Pass $indexParams for a full-text or invert index: + * + * $schema->addString('body', indexParams: ZVecIndexParams::forFts()); + * + * That is an alternative to creating the collection and then calling + * createIndex('body', ...). Upstream validates a mismatched index type at + * ZVec::create() time, so passing vector params fails there with a message + * naming the field. + * + * @throws ZVecException On FFI error, duplicate field, or when both + * $withInvertIndex and $indexParams are given + */ + public function addString( + string $name, + bool $nullable = false, + bool $withInvertIndex = false, + ?ZVecIndexParams $indexParams = null, + ): self { + if ($indexParams === null) { + self::ffi()->zvec_schema_add_field_string($this->handle, $name, $nullable ? 1 : 0, $withInvertIndex ? 1 : 0); + return $this; + } + if ($withInvertIndex) { + throw new ZVecException('Use either $withInvertIndex or $indexParams, not both'); + } + ZVec::checkStatus(self::ffi()->zvec_schema_add_field_string_with_index( + $this->handle, + $name, + $nullable ? 1 : 0, + $indexParams->getHandle(), + )); return $this; } diff --git a/tests/test_ffi_load.phpt b/tests/test_ffi_load.phpt index ff62c65..e149314 100644 --- a/tests/test_ffi_load.phpt +++ b/tests/test_ffi_load.phpt @@ -14,6 +14,7 @@ $requiredFunctions = [ 'zvec_init', 'zvec_schema_create', 'zvec_schema_free', + 'zvec_schema_add_field_string_with_index', 'zvec_collection_create', 'zvec_collection_open', 'zvec_collection_free', @@ -141,7 +142,7 @@ try { echo "DONE\n"; ?> --EXPECT-- -All 69 FFI symbols resolved successfully +All 70 FFI symbols resolved successfully No FFI::cdef() inline string found in src/ZVec.php Header file zvec_ffi_php.h is used as source of truth Basic create/insert/optimize works diff --git a/tests/test_schema_fts_index.phpt b/tests/test_schema_fts_index.phpt new file mode 100644 index 0000000..d064cf4 --- /dev/null +++ b/tests/test_schema_fts_index.phpt @@ -0,0 +1,122 @@ +--TEST-- +Schema: declare an FTS or invert index on a STRING field at create time +--SKIPIF-- + +--FILE-- +setTopk(10)->setFts('body', $term); + $pks = array_map(static fn(ZVecDoc $d): string => $d->getPk(), $collection->queryVector($query)); + sort($pks); + return $pks; +} + +$path = __DIR__ . '/../test_dbs/schema_fts_' . uniqid(); +$invertPath = __DIR__ . '/../test_dbs/schema_invert_' . uniqid(); +$hnswPath = __DIR__ . '/../test_dbs/schema_hnsw_' . uniqid(); + +try { + $schema = new ZVecSchema('schema_fts_test'); + $schema->addString('body', indexParams: ZVecIndexParams::forFts()) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + + $collection = ZVec::create($path, $schema); + + foreach ([ + ['d1', 'the quick brown fox jumps'], + ['d2', 'a fast brown rabbit'], + ['d3', 'completely unrelated text about databases'], + ] as [$pk, $body]) { + $collection->insert( + (new ZVecDoc($pk))->setString('body', $body)->setVectorFp32('vec', [1.0, 0.0, 0.0, 0.0]) + ); + } + $collection->flush(); + + // The index exists from the first insert: no createIndex() call anywhere. + echo 'fts in schema: ' . implode(',', ftsPks($collection, 'fox')) . "\n"; + echo 'index type: ' . $collection->getFieldSchema('body')->getIndexType() . "\n"; + + $collection->close(); + $collection = ZVec::open($path); + echo 'after reopen: ' . implode(',', ftsPks($collection, 'fox')) . "\n"; + echo 'index type after reopen: ' . $collection->getFieldSchema('body')->getIndexType() . "\n"; + + // Invert index declared the same way. + $invertSchema = new ZVecSchema('schema_invert_test'); + $invertSchema->addString('tag', indexParams: ZVecIndexParams::forInvert()) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $invertCollection = ZVec::create($invertPath, $invertSchema); + echo 'invert type: ' . $invertCollection->getFieldSchema('tag')->getIndexType() . "\n"; + + // Requesting both an invert flag and explicit params is ambiguous. + try { + (new ZVecSchema('schema_conflict'))->addString( + 'a', + withInvertIndex: true, + indexParams: ZVecIndexParams::forFts(), + ); + echo "both flags: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo 'both flags: ' . $e->getMessage() . "\n"; + } + + // A duplicate field is reported here, unlike the other add*() methods which + // drop the Status from add_field(). + try { + $duplicate = new ZVecSchema('schema_duplicate'); + $duplicate->addString('body'); + $duplicate->addString('body', indexParams: ZVecIndexParams::forFts()); + echo "duplicate: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo 'duplicate: ' . (str_contains($e->getMessage(), 'already exists') ? 'rejected' : $e->getMessage()) . "\n"; + } + + // Vector index params on a scalar field are caught by upstream at create + // time, which names the field and the type. + $hnswSchema = new ZVecSchema('schema_hnsw_string'); + $hnswSchema->addString('body', indexParams: ZVecIndexParams::forHnsw(ZVecSchema::METRIC_IP)) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + try { + ZVec::create($hnswPath, $hnswSchema); + echo "hnsw on string: NOT REJECTED\n"; + } catch (ZVecException $e) { + echo 'hnsw on string: ' . (str_contains($e->getMessage(), 'does not support vector index params') + ? 'rejected' : $e->getMessage()) . "\n"; + } + + // Backward compatibility of the pre-existing argument combinations. + $bcSchema = new ZVecSchema('schema_bc'); + $bcSchema->addString('x') + ->addString('x2', withInvertIndex: true) + ->addVectorFp32('vec', dimension: 4, metricType: ZVecSchema::METRIC_IP); + $bcPath = __DIR__ . '/../test_dbs/schema_bc_' . uniqid(); + $bcCollection = ZVec::create($bcPath, $bcSchema); + echo 'bc: ' . ($bcCollection->getFieldSchema('x2')->hasInvertIndex() ? 'ok' : 'FAILED') . "\n"; + $bcCollection->close(); + exec('rm -rf ' . escapeshellarg($bcPath)); + + echo "PASS\n"; +} finally { + exec('rm -rf ' . escapeshellarg($path)); + exec('rm -rf ' . escapeshellarg($invertPath)); + exec('rm -rf ' . escapeshellarg($hnswPath)); +} +?> +--EXPECT-- +fts in schema: d1 +index type: 11 +after reopen: d1 +index type after reopen: 11 +invert type: 10 +both flags: Use either $withInvertIndex or $indexParams, not both +duplicate: rejected +hnsw on string: rejected +bc: ok +PASS