diff --git a/tests/e2e_tests/test_data/compatibility/populate_db.py b/tests/e2e_tests/test_data/compatibility/populate_db.py index 1564db9e25..3bbee03314 100755 --- a/tests/e2e_tests/test_data/compatibility/populate_db.py +++ b/tests/e2e_tests/test_data/compatibility/populate_db.py @@ -21,6 +21,11 @@ def drop_collection(name: str): def create_collection(name: str, memmap_threshold_kb: int, on_disk: bool, datatype: str, quantization_config: Optional[dict] = None): + # `turbo4` is a dense-only storage datatype, sparse vector configs reject it + sparse_index = {"on_disk": on_disk} + if datatype != "turbo4": + sparse_index["datatype"] = datatype + # create collection with a lower `indexing_threshold_kb` to generate the HNSW index response = requests.put( f"http://{QDRANT_HOST}/collections/{name}", @@ -45,10 +50,7 @@ def create_collection(name: str, memmap_threshold_kb: int, on_disk: bool, dataty }, "sparse_vectors": { "text": { - "index": { - "on_disk": on_disk, - "datatype": datatype, - } + "index": sparse_index, } }, "optimizers_config": { @@ -65,14 +67,15 @@ def create_collection(name: str, memmap_threshold_kb: int, on_disk: bool, dataty def create_payload_indexes(name: str, on_disk_payload_index: bool): - # keyword + # keyword, with the extra on-disk structure backing `match: {"prefix": ...}` response = requests.put( f"http://{QDRANT_HOST}/collections/{name}/index", json={ "field_name": "keyword_field", "field_schema": { "type": "keyword", - "on_disk": on_disk_payload_index + "on_disk": on_disk_payload_index, + "prefix": True } }, ) @@ -361,6 +364,9 @@ if __name__ == "__main__": populate_collection("test_collection_product_x16", on_disk=False, quantization_config={"product": {"compression": "x16"}}) populate_collection("test_collection_product_x8", on_disk=False, quantization_config={"product": {"compression": "x8"}}) populate_collection("test_collection_binary", on_disk=False, quantization_config={"binary": {"always_ram": True}}) + populate_collection("test_collection_turbo_bits1_5", on_disk=False, quantization_config={"turbo": {"bits": "bits1_5"}}) + populate_collection("test_collection_turbo_bits4", on_disk=False, quantization_config={"turbo": {"bits": "bits4"}}) populate_collection("test_collection_mmap_field_index", on_disk=True, on_disk_payload_index=True) populate_collection("test_collection_vector_datatype_u8", on_disk=True, datatype="uint8") populate_collection("test_collection_vector_datatype_f16", on_disk=True, datatype="float16") + populate_collection("test_collection_vector_datatype_turbo4", on_disk=True, datatype="turbo4") diff --git a/tests/e2e_tests/test_data_compatibility.py b/tests/e2e_tests/test_data_compatibility.py index 821f52a235..9a4eca761a 100644 --- a/tests/e2e_tests/test_data_compatibility.py +++ b/tests/e2e_tests/test_data_compatibility.py @@ -21,6 +21,7 @@ class TestStorageCompatibility: """ VERSIONS = [ + "v1.19.0", "v1.18.1", "v1.18.0", "v1.17.1", @@ -31,7 +32,8 @@ class TestStorageCompatibility: "v1.16.0", ] - EXPECTED_COLLECTIONS = [ + # Collections present in every published compatibility archive + BASE_COLLECTIONS = [ "test_collection_vector_memory", "test_collection_vector_on_disk", "test_collection_vector_on_disk_threshold", @@ -46,6 +48,36 @@ class TestStorageCompatibility: "test_collection_vector_datatype_f16" ] + # Collections added to `populate_db.py` later on, keyed by the oldest release + # whose published archive contains them. Archives are generated once per + # release, so older ones keep the collection set of their own generation. + # + # TurboQuant quantization is supported since v1.18.0 and the `turbo4` datatype + # since v1.18.3, but the generator only started creating these collections for + # v1.19.0. Lower the key if an older archive is regenerated. + VERSIONED_COLLECTIONS = { + (1, 19, 0): [ + "test_collection_turbo_bits1_5", + "test_collection_turbo_bits4", + "test_collection_vector_datatype_turbo4", + ], + } + + @staticmethod + def _parse_version(version: str) -> tuple[int, ...]: + """Turn a version tag such as "v1.19.1" into a comparable tuple.""" + return tuple(int(part) for part in version.lstrip("v").split(".")) + + @classmethod + def _expected_collections(cls, version: str) -> list[str]: + """Collections the archive of `version` is expected to contain.""" + parsed = cls._parse_version(version) + collections = list(cls.BASE_COLLECTIONS) + for min_version, names in cls.VERSIONED_COLLECTIONS.items(): + if parsed >= min_version: + collections.extend(names) + return collections + @staticmethod def _download_compatibility_data(version: str, storage_test_dir: Path) -> Path: """Download compatibility data for a specific version. @@ -72,7 +104,16 @@ class TestStorageCompatibility: for chunk in response.iter_content(chunk_size=1024 * 1024): if chunk: f.write(chunk) + except requests.exceptions.HTTPError as e: + # A missing archive means the version was never published, which no + # amount of retrying fixes. Skipping it would report the version as + # covered while nothing ran, so fail loudly instead. + if e.response is not None and e.response.status_code == 404: + pytest.fail(f"No published compatibility archive for {version}: {url}") + pytest.skip(f"Could not download compatibility data for {version}: {e}") except requests.exceptions.RequestException as e: + # Connection resets, timeouts and the like are transient, do not turn + # unrelated pull requests red over them pytest.skip(f"Could not download compatibility data for {version}: {e}") return compatibility_file @@ -103,7 +144,7 @@ class TestStorageCompatibility: DENSE_DIM = 256 MULTI_DENSE_DIM = 128 - def _check_collections(self, host: str, port: int) -> tuple[bool, str]: + def _check_collections(self, host: str, port: int, version: str) -> tuple[bool, str]: """Check that all collections are loaded properly. Returns: @@ -116,7 +157,7 @@ class TestStorageCompatibility: except Exception as e: return False, f"Error listing collections: {e}" - expected = set(self.EXPECTED_COLLECTIONS) + expected = set(self._expected_collections(version)) found = set(collections) missing = expected - found if missing: @@ -132,7 +173,7 @@ class TestStorageCompatibility: return True, "" - def _query_collections(self, host: str, port: int) -> tuple[bool, str]: + def _query_collections(self, host: str, port: int, version: str) -> tuple[bool, str]: """Run queries against all collections to verify data is actually accessible. Sends one query of each kind per collection: dense, sparse, and multivector @@ -140,7 +181,7 @@ class TestStorageCompatibility: """ base_url = f"http://{host}:{port}" - for collection in self.EXPECTED_COLLECTIONS: + for collection in self._expected_collections(version): try: # Dense vector search resp = requests.post( @@ -185,6 +226,18 @@ class TestStorageCompatibility: if not resp.ok: return False, f"Keyword filter scroll failed on {collection}: {resp.status_code} {resp.text}" + # Scroll with keyword prefix filter, archives without the `prefix` + # index option answer it by scanning + resp = requests.post( + f"{base_url}/collections/{collection}/points/scroll", + json={ + "filter": {"must": [{"key": "keyword_field", "match": {"prefix": "hel"}}]}, + "limit": 3, + }, + ) + if not resp.ok: + return False, f"Keyword prefix filter scroll failed on {collection}: {resp.status_code} {resp.text}" + # Scroll with float range filter resp = requests.post( f"{base_url}/collections/{collection}/points/scroll", @@ -303,11 +356,11 @@ class TestStorageCompatibility: if not client.wait_for_server(): return False, f"Server failed to start for {test_name} test ({version})" - success, error_msg = self._check_collections(container_info.host, container_info.http_port) + success, error_msg = self._check_collections(container_info.host, container_info.http_port, version) if not success: return False, f"{test_name.capitalize()} compatibility failed for {version}: {error_msg}" - success, error_msg = self._query_collections(container_info.host, container_info.http_port) + success, error_msg = self._query_collections(container_info.host, container_info.http_port, version) if not success: return False, f"{test_name.capitalize()} query verification failed for {version}: {error_msg}"