mirror of
https://github.com/qdrant/qdrant.git
synced 2026-09-28 00:47:32 -05:00
test: cover TurboQuant, turbo4 datatype and keyword prefix in compat data (#10165)
* test: cover TurboQuant, turbo4 datatype and keyword prefix in compat data Extend the storage compatibility fixture with vector and payload index features that landed since the generator was last updated: * TurboQuant quantization, one collection per persisted blob layout (bits1_5 and the default bits4) * the turbo4 storage datatype, on both dense and multivector storage * the keyword index `prefix` option, plus a matching prefix scroll in the query battery Sparse vector configs reject the turbo4 datatype, so create_collection omits the sparse datatype for that collection rather than forwarding it. This is a no-op for every other collection. Archives are generated once per release and keep the collection set of their own generation, so expected collections are now resolved per version and the new ones are only required from v1.19.0 onward. The prefix scroll stays ungated: archives without the prefix index answer it by scanning. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> * test: fail instead of skip when a compatibility archive is missing A 404 from the compatibility bucket means the archive was never published, which no amount of retrying fixes. Skipping it reported the version as covered while nothing ran, so a pull request adding a version could stay green with its new coverage never executing. Fail on 404 and keep skipping connection resets and timeouts, so a bucket outage still does not turn unrelated pull requests red. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5
parent
1a8c9a6397
commit
00a0e0d768
@@ -21,6 +21,11 @@ def drop_collection(name: str):
|
||||
|
||||
|
||||
def create_collection(name: str, memmap_threshold_kb: int, on_disk: bool, datatype: str, quantization_config: Optional[dict] = None):
|
||||
# `turbo4` is a dense-only storage datatype, sparse vector configs reject it
|
||||
sparse_index = {"on_disk": on_disk}
|
||||
if datatype != "turbo4":
|
||||
sparse_index["datatype"] = datatype
|
||||
|
||||
# create collection with a lower `indexing_threshold_kb` to generate the HNSW index
|
||||
response = requests.put(
|
||||
f"http://{QDRANT_HOST}/collections/{name}",
|
||||
@@ -45,10 +50,7 @@ def create_collection(name: str, memmap_threshold_kb: int, on_disk: bool, dataty
|
||||
},
|
||||
"sparse_vectors": {
|
||||
"text": {
|
||||
"index": {
|
||||
"on_disk": on_disk,
|
||||
"datatype": datatype,
|
||||
}
|
||||
"index": sparse_index,
|
||||
}
|
||||
},
|
||||
"optimizers_config": {
|
||||
@@ -65,14 +67,15 @@ def create_collection(name: str, memmap_threshold_kb: int, on_disk: bool, dataty
|
||||
|
||||
|
||||
def create_payload_indexes(name: str, on_disk_payload_index: bool):
|
||||
# keyword
|
||||
# keyword, with the extra on-disk structure backing `match: {"prefix": ...}`
|
||||
response = requests.put(
|
||||
f"http://{QDRANT_HOST}/collections/{name}/index",
|
||||
json={
|
||||
"field_name": "keyword_field",
|
||||
"field_schema": {
|
||||
"type": "keyword",
|
||||
"on_disk": on_disk_payload_index
|
||||
"on_disk": on_disk_payload_index,
|
||||
"prefix": True
|
||||
}
|
||||
},
|
||||
)
|
||||
@@ -361,6 +364,9 @@ if __name__ == "__main__":
|
||||
populate_collection("test_collection_product_x16", on_disk=False, quantization_config={"product": {"compression": "x16"}})
|
||||
populate_collection("test_collection_product_x8", on_disk=False, quantization_config={"product": {"compression": "x8"}})
|
||||
populate_collection("test_collection_binary", on_disk=False, quantization_config={"binary": {"always_ram": True}})
|
||||
populate_collection("test_collection_turbo_bits1_5", on_disk=False, quantization_config={"turbo": {"bits": "bits1_5"}})
|
||||
populate_collection("test_collection_turbo_bits4", on_disk=False, quantization_config={"turbo": {"bits": "bits4"}})
|
||||
populate_collection("test_collection_mmap_field_index", on_disk=True, on_disk_payload_index=True)
|
||||
populate_collection("test_collection_vector_datatype_u8", on_disk=True, datatype="uint8")
|
||||
populate_collection("test_collection_vector_datatype_f16", on_disk=True, datatype="float16")
|
||||
populate_collection("test_collection_vector_datatype_turbo4", on_disk=True, datatype="turbo4")
|
||||
|
||||
@@ -21,6 +21,7 @@ class TestStorageCompatibility:
|
||||
"""
|
||||
|
||||
VERSIONS = [
|
||||
"v1.19.0",
|
||||
"v1.18.1",
|
||||
"v1.18.0",
|
||||
"v1.17.1",
|
||||
@@ -31,7 +32,8 @@ class TestStorageCompatibility:
|
||||
"v1.16.0",
|
||||
]
|
||||
|
||||
EXPECTED_COLLECTIONS = [
|
||||
# Collections present in every published compatibility archive
|
||||
BASE_COLLECTIONS = [
|
||||
"test_collection_vector_memory",
|
||||
"test_collection_vector_on_disk",
|
||||
"test_collection_vector_on_disk_threshold",
|
||||
@@ -46,6 +48,36 @@ class TestStorageCompatibility:
|
||||
"test_collection_vector_datatype_f16"
|
||||
]
|
||||
|
||||
# Collections added to `populate_db.py` later on, keyed by the oldest release
|
||||
# whose published archive contains them. Archives are generated once per
|
||||
# release, so older ones keep the collection set of their own generation.
|
||||
#
|
||||
# TurboQuant quantization is supported since v1.18.0 and the `turbo4` datatype
|
||||
# since v1.18.3, but the generator only started creating these collections for
|
||||
# v1.19.0. Lower the key if an older archive is regenerated.
|
||||
VERSIONED_COLLECTIONS = {
|
||||
(1, 19, 0): [
|
||||
"test_collection_turbo_bits1_5",
|
||||
"test_collection_turbo_bits4",
|
||||
"test_collection_vector_datatype_turbo4",
|
||||
],
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _parse_version(version: str) -> tuple[int, ...]:
|
||||
"""Turn a version tag such as "v1.19.1" into a comparable tuple."""
|
||||
return tuple(int(part) for part in version.lstrip("v").split("."))
|
||||
|
||||
@classmethod
|
||||
def _expected_collections(cls, version: str) -> list[str]:
|
||||
"""Collections the archive of `version` is expected to contain."""
|
||||
parsed = cls._parse_version(version)
|
||||
collections = list(cls.BASE_COLLECTIONS)
|
||||
for min_version, names in cls.VERSIONED_COLLECTIONS.items():
|
||||
if parsed >= min_version:
|
||||
collections.extend(names)
|
||||
return collections
|
||||
|
||||
@staticmethod
|
||||
def _download_compatibility_data(version: str, storage_test_dir: Path) -> Path:
|
||||
"""Download compatibility data for a specific version.
|
||||
@@ -72,7 +104,16 @@ class TestStorageCompatibility:
|
||||
for chunk in response.iter_content(chunk_size=1024 * 1024):
|
||||
if chunk:
|
||||
f.write(chunk)
|
||||
except requests.exceptions.HTTPError as e:
|
||||
# A missing archive means the version was never published, which no
|
||||
# amount of retrying fixes. Skipping it would report the version as
|
||||
# covered while nothing ran, so fail loudly instead.
|
||||
if e.response is not None and e.response.status_code == 404:
|
||||
pytest.fail(f"No published compatibility archive for {version}: {url}")
|
||||
pytest.skip(f"Could not download compatibility data for {version}: {e}")
|
||||
except requests.exceptions.RequestException as e:
|
||||
# Connection resets, timeouts and the like are transient, do not turn
|
||||
# unrelated pull requests red over them
|
||||
pytest.skip(f"Could not download compatibility data for {version}: {e}")
|
||||
|
||||
return compatibility_file
|
||||
@@ -103,7 +144,7 @@ class TestStorageCompatibility:
|
||||
DENSE_DIM = 256
|
||||
MULTI_DENSE_DIM = 128
|
||||
|
||||
def _check_collections(self, host: str, port: int) -> tuple[bool, str]:
|
||||
def _check_collections(self, host: str, port: int, version: str) -> tuple[bool, str]:
|
||||
"""Check that all collections are loaded properly.
|
||||
|
||||
Returns:
|
||||
@@ -116,7 +157,7 @@ class TestStorageCompatibility:
|
||||
except Exception as e:
|
||||
return False, f"Error listing collections: {e}"
|
||||
|
||||
expected = set(self.EXPECTED_COLLECTIONS)
|
||||
expected = set(self._expected_collections(version))
|
||||
found = set(collections)
|
||||
missing = expected - found
|
||||
if missing:
|
||||
@@ -132,7 +173,7 @@ class TestStorageCompatibility:
|
||||
|
||||
return True, ""
|
||||
|
||||
def _query_collections(self, host: str, port: int) -> tuple[bool, str]:
|
||||
def _query_collections(self, host: str, port: int, version: str) -> tuple[bool, str]:
|
||||
"""Run queries against all collections to verify data is actually accessible.
|
||||
|
||||
Sends one query of each kind per collection: dense, sparse, and multivector
|
||||
@@ -140,7 +181,7 @@ class TestStorageCompatibility:
|
||||
"""
|
||||
base_url = f"http://{host}:{port}"
|
||||
|
||||
for collection in self.EXPECTED_COLLECTIONS:
|
||||
for collection in self._expected_collections(version):
|
||||
try:
|
||||
# Dense vector search
|
||||
resp = requests.post(
|
||||
@@ -185,6 +226,18 @@ class TestStorageCompatibility:
|
||||
if not resp.ok:
|
||||
return False, f"Keyword filter scroll failed on {collection}: {resp.status_code} {resp.text}"
|
||||
|
||||
# Scroll with keyword prefix filter, archives without the `prefix`
|
||||
# index option answer it by scanning
|
||||
resp = requests.post(
|
||||
f"{base_url}/collections/{collection}/points/scroll",
|
||||
json={
|
||||
"filter": {"must": [{"key": "keyword_field", "match": {"prefix": "hel"}}]},
|
||||
"limit": 3,
|
||||
},
|
||||
)
|
||||
if not resp.ok:
|
||||
return False, f"Keyword prefix filter scroll failed on {collection}: {resp.status_code} {resp.text}"
|
||||
|
||||
# Scroll with float range filter
|
||||
resp = requests.post(
|
||||
f"{base_url}/collections/{collection}/points/scroll",
|
||||
@@ -303,11 +356,11 @@ class TestStorageCompatibility:
|
||||
if not client.wait_for_server():
|
||||
return False, f"Server failed to start for {test_name} test ({version})"
|
||||
|
||||
success, error_msg = self._check_collections(container_info.host, container_info.http_port)
|
||||
success, error_msg = self._check_collections(container_info.host, container_info.http_port, version)
|
||||
if not success:
|
||||
return False, f"{test_name.capitalize()} compatibility failed for {version}: {error_msg}"
|
||||
|
||||
success, error_msg = self._query_collections(container_info.host, container_info.http_port)
|
||||
success, error_msg = self._query_collections(container_info.host, container_info.http_port, version)
|
||||
if not success:
|
||||
return False, f"{test_name.capitalize()} query verification failed for {version}: {error_msg}"
|
||||
|
||||
|
||||
Reference in New Issue
Block a user