diff --git a/README.md b/README.md index 0ffa98f0e..bc44b36d9 100644 --- a/README.md +++ b/README.md @@ -87,6 +87,7 @@ All the database client supported | tencent_es | `pip install vectordb-bench[tencent_es]` | | alisql | `pip install vectordb-bench[alisql]` | | polardb | `pip install vectordb-bench[polardb]` | +| polardb_pg | `pip install vectordb-bench[pgvector]` | | doris | `pip install vectordb-bench[doris]` | | zvec | `pip install vectordb-bench[zvec]` | | endee | `pip install vectordb-bench[endee]` | @@ -116,6 +117,7 @@ Options: --help Show this message and exit. Commands: + polardbpghnsw pgvectorhnsw pgvectorivfflat vectorchordrq @@ -263,6 +265,47 @@ Options: --help Show this message and exit. ``` +### Run polardb_pg from command line + +polardb_pg supports HNSW with PQ, SQ4, SQ8, RaBitQ, and Graph Cache. +This PostgreSQL client is separate from the existing MySQL-compatible PolarDB +commands and reuses the `pgvector` dependency extra: + +```shell +pip install 'vectordb-bench[pgvector]' +``` + +PolarDB-specific internal quantization and Graph Cache options require PolarDB +vector extension 0.8.3.1 or later. Use `--skip-graph-cache` when Graph Cache is +not configured on the server. + +**Example: Run hnsw index test** + +```shell +vectordbbench polardbpghnsw \ + --case-type Performance1024D1M \ + --db-label polardb-pg-rabitq8 \ + --user-name postgres --password '' \ + --host localhost --port 5432 --db-name vectordb \ + --m 16 --ef-construction 256 --ef-search 200 \ + --quantization rabitq --quantization-nbits 8 \ + --iterative-scan off --graph-cache +``` + +To list the options for polardb_pg, execute +`vectordbbench polardbpghnsw --help`. The following are some PolarDB-specific +command-line options. + +```text + --quantization [pq|sq4|sq8|rabitq] + --pq-m INTEGER + --train-samples INTEGER + --quantization-nbits [1|4|8] + --graph-cache / --skip-graph-cache + --graph-cache-timeout INTEGER + --iterative-scan [off|strict_order|relaxed_order] +``` + ### Run VectorChord (vchordrq) from command line VectorChord is a PostgreSQL extension for scalable vector similarity search using IVF + RaBitQ indexing. diff --git a/pyproject.toml b/pyproject.toml index 9fbbb7748..43f793c60 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -54,6 +54,9 @@ test = [ "black", "ruff", "pytest", + "psycopg", + "psycopg-binary", + "pgvector", ] restful = [ "flask" ] qdrant = [ "qdrant-client" ] @@ -63,7 +66,7 @@ elastic = [ "elasticsearch" ] # For elastic and aliyun_elasticsearch pgvector = [ "psycopg", "psycopg-binary", "pgvector" ] -# for pgvector, pgvectorscale, pgdiskann, alloydb, and lakebase_vector +# for pgvector, pgvectorscale, pgdiskann, alloydb, lakebase_vector, and PolarDB for PostgreSQL HNSW pgvecto_rs = [ "pgvecto_rs[psycopg3]>=0.2.2" ] redis = [ "redis" ] diff --git a/tests/test_polardb_pg.py b/tests/test_polardb_pg.py new file mode 100644 index 000000000..4d8fdc58b --- /dev/null +++ b/tests/test_polardb_pg.py @@ -0,0 +1,473 @@ +from __future__ import annotations + +from contextlib import contextmanager +from typing import TYPE_CHECKING +from unittest.mock import MagicMock, call + +import pytest +from click.testing import CliRunner +from pydantic import ValidationError + +from vectordb_bench.backend.cases import CaseLabel +from vectordb_bench.backend.clients import DB +from vectordb_bench.backend.clients.api import IndexType +from vectordb_bench.backend.clients.pgvector.pgvector import PgVector +from vectordb_bench.backend.clients.polardb_pg.cli import PolarDBPgHNSW as PolarDBPgHNSWCommand +from vectordb_bench.backend.clients.polardb_pg.config import PolarDBPgHNSWConfig +from vectordb_bench.backend.clients.polardb_pg.polardb_pg import ( + HNSWGraphCacheDetail, + PolarDBPgHNSW, +) +from vectordb_bench.frontend.config.dbCaseConfigs import get_case_config_inputs +from vectordb_bench.models import CaseConfigParamType + +if TYPE_CHECKING: + from collections.abc import Iterator + + +def make_config(**kwargs) -> PolarDBPgHNSWConfig: + return PolarDBPgHNSWConfig( + metric_type="COSINE", + m=16, + ef_construction=256, + ef_search=200, + **kwargs, + ) + + +def cache_detail( + status: str, + *, + requested: str = "on", + usable: bool = False, + last_error: str | None = None, +) -> HNSWGraphCacheDetail: + detail = f"requested={requested} status={status} usable={'yes' if usable else 'no'}" + if last_error is not None: + detail += f" last_error={last_error}" + return HNSWGraphCacheDetail.parse(detail) + + +def make_client(details: list[HNSWGraphCacheDetail], **config_kwargs) -> PolarDBPgHNSW: + client = object.__new__(PolarDBPgHNSW) + client.case_config = make_config(**config_kwargs) + client._index_name = "pgvector_index" + client._check_graph_cache_role = MagicMock() + client._graph_cache_detail = MagicMock(side_effect=details) + client._call_graph_cache_function = MagicMock(return_value=True) + return client + + +class TestPolarDBPgConfig: + def test_rabitq_graph_cache_index_options(self): + config = make_config( + quantization="rabitq", + quantization_nbits=8, + train_samples=100000, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert options == { + "m": "16", + "ef_construction": "256", + "quantization": "rabitq", + "train_samples": "100000", + "quantization_nbits": "8", + "cache": "on", + } + + def test_frontend_quantization_key_maps_to_index_option(self): + config = make_config( + hnsw_quantization="rabitq", + quantization_nbits=8, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert config.quantization == "rabitq" + assert options["quantization"] == "rabitq" + assert options["quantization_nbits"] == "8" + + def test_pq_index_options(self): + config = make_config( + graph_cache=False, + quantization="pq", + pq_m=32, + train_samples=5000, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert options == { + "m": "16", + "ef_construction": "256", + "quantization": "pq", + "pq_m": "32", + "train_samples": "5000", + } + + @pytest.mark.parametrize("quantization", ["sq4", "sq8"]) + def test_sq_index_options(self, quantization: str): + config = make_config( + graph_cache=False, + quantization=quantization, + train_samples=5000, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert options == { + "m": "16", + "ef_construction": "256", + "quantization": quantization, + "train_samples": "5000", + } + + def test_graph_cache_can_be_disabled(self): + config = make_config(graph_cache=False, iterative_scan="relaxed_order") + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + assert "cache" not in options + + def test_graph_cache_timeout_must_be_positive(self): + assert make_config().graph_cache_timeout == 3600 + with pytest.raises(ValidationError): + make_config(graph_cache_timeout=0) + + @pytest.mark.parametrize("nbits", [1, 4, 8]) + def test_supported_rabitq_nbits(self, nbits: int): + assert make_config(quantization="rabitq", quantization_nbits=nbits).quantization_nbits == nbits + + @pytest.mark.parametrize("nbits", [0, 2, 9]) + def test_rejects_unsupported_rabitq_nbits(self, nbits: int): + with pytest.raises(ValidationError): + make_config(quantization="rabitq", quantization_nbits=nbits) + + def test_rejects_iterative_scan_with_graph_cache(self): + with pytest.raises(ValidationError, match="Graph Cache requires"): + make_config(iterative_scan="strict_order") + + def test_frontend_post_load_switch_controls_index_timing(self): + before_load = make_config(post_load_index=False) + after_load = make_config(post_load_index=True) + + assert before_load.create_index_before_load is True + assert before_load.create_index_after_load is False + assert after_load.create_index_before_load is False + assert after_load.create_index_after_load is True + + def test_rejects_quantizer_specific_options(self): + with pytest.raises(ValidationError, match="pq_m"): + make_config(quantization="sq8", pq_m=16) + with pytest.raises(ValidationError, match="quantization_nbits"): + make_config(quantization="pq", quantization_nbits=8) + + def test_rejects_opq(self): + with pytest.raises(ValidationError): + make_config(quantization="opq") + + def test_connection_password_is_redacted(self, monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture): + connection = MagicMock() + cursor = MagicMock() + monkeypatch.setattr(PgVector, "_create_connection", MagicMock(return_value=(connection, cursor))) + caplog.set_level("INFO", logger="vectordb_bench.backend.clients.pgvector.pgvector") + + PgVector( + dim=8, + db_config={ + "connect_config": { + "host": "localhost", + "port": 5432, + "dbname": "vectordb", + "user": "postgres", + "password": "do-not-log-this-password", + }, + "table_name": "vdbbench_table_test", + }, + db_case_config=make_config(graph_cache=False), + ) + + assert "do-not-log-this-password" not in caplog.text + assert "**********" in caplog.text + + def test_polardb_name_is_used_during_base_initialization( + self, + monkeypatch: pytest.MonkeyPatch, + caplog: pytest.LogCaptureFixture, + ): + connection = MagicMock() + cursor = MagicMock() + monkeypatch.setattr(PgVector, "_create_connection", MagicMock(return_value=(connection, cursor))) + caplog.set_level("INFO", logger="vectordb_bench.backend.clients.pgvector.pgvector") + + client = PolarDBPgHNSW( + dim=8, + db_config={ + "connect_config": { + "host": "localhost", + "port": 5432, + "dbname": "vectordb", + "user": "postgres", + "password": "secret", + }, + "table_name": "vdbbench_table_test", + }, + db_case_config=make_config(graph_cache=False), + ) + + assert client.name == "PolarDBPG" + assert "PolarDBPG config values" in caplog.text + assert "PgVector config values" not in caplog.text + + +class TestGraphCacheDetail: + def test_parse_ready_detail(self): + detail = HNSWGraphCacheDetail.parse( + "requested=on dbid=1 index_oid=2 status=ready usable=yes graph_bytes=100 last_error=none", + ) + assert detail.requested == "on" + assert detail.status == "ready" + assert detail.usable is True + assert detail.last_error is None + + def test_parse_error_with_spaces(self): + detail = HNSWGraphCacheDetail.parse( + "requested=on status=empty usable=no last_error=could not read graph page", + ) + assert detail.last_error == "could not read graph page" + + def test_rejects_incomplete_detail(self): + with pytest.raises(RuntimeError, match="missing usable"): + HNSWGraphCacheDetail.parse("requested=on status=empty") + + +class TestGraphCacheLifecycle: + def test_search_only_initialization_waits_for_cache(self, monkeypatch: pytest.MonkeyPatch): + def base_init( + client: PolarDBPgHNSW, + *_args: object, + db_case_config: PolarDBPgHNSWConfig, + **_kwargs: object, + ) -> None: + client.case_config = db_case_config + client._index_name = "pgvector_index" + + @contextmanager + def fake_init(_client: PolarDBPgHNSW) -> Iterator[None]: + yield + + wait_for_cache = MagicMock() + monkeypatch.setattr(PgVector, "__init__", base_init) + monkeypatch.setattr(PolarDBPgHNSW, "init", fake_init) + monkeypatch.setattr(PolarDBPgHNSW, "_ensure_graph_cache_ready", wait_for_cache) + + PolarDBPgHNSW(db_case_config=make_config(), drop_old=False) + + wait_for_cache.assert_called_once_with() + + def test_post_insert_waits_after_index_creation(self, monkeypatch: pytest.MonkeyPatch): + create_index = MagicMock() + wait_for_cache = MagicMock() + client = object.__new__(PolarDBPgHNSW) + client.case_config = make_config() + monkeypatch.setattr(PgVector, "_post_insert", create_index) + monkeypatch.setattr(PolarDBPgHNSW, "_ensure_graph_cache_ready", wait_for_cache) + + client._post_insert() + + create_index.assert_called_once_with() + wait_for_cache.assert_called_once_with() + + def test_ready_cache_needs_no_lifecycle_call(self): + client = make_client([cache_detail("ready", usable=True)]) + + client._ensure_graph_cache_ready() + + client._check_graph_cache_role.assert_called_once_with() + client._call_graph_cache_function.assert_not_called() + + def test_empty_cache_is_scheduled_before_ready(self, monkeypatch: pytest.MonkeyPatch): + client = make_client( + [ + cache_detail("empty"), + cache_detail("building"), + cache_detail("ready", usable=True), + ], + ) + monkeypatch.setattr("vectordb_bench.backend.clients.polardb_pg.polardb_pg.time.sleep", lambda _: None) + + client._ensure_graph_cache_ready() + + client._call_graph_cache_function.assert_called_once_with("hnsw_schedule_cache_rebuild") + + def test_stale_cache_is_released_and_rebuilt(self, monkeypatch: pytest.MonkeyPatch): + client = make_client( + [ + cache_detail("ready", usable=False), + cache_detail("draining"), + cache_detail("empty"), + cache_detail("building"), + cache_detail("ready", usable=True), + ], + ) + monkeypatch.setattr("vectordb_bench.backend.clients.polardb_pg.polardb_pg.time.sleep", lambda _: None) + + client._ensure_graph_cache_ready() + + assert client._call_graph_cache_function.call_args_list == [ + call("hnsw_release_cache"), + call("hnsw_schedule_cache_rebuild"), + ] + + def test_not_preloaded_fails_immediately(self): + client = make_client([cache_detail("not_preloaded")]) + with pytest.raises(RuntimeError, match="shared_preload_libraries"): + client._ensure_graph_cache_ready() + + def test_index_without_cache_reloption_fails(self): + client = make_client([cache_detail("empty", requested="off")]) + with pytest.raises(RuntimeError, match="cache=on"): + client._ensure_graph_cache_ready() + + def test_build_error_fails_without_retry(self): + client = make_client([cache_detail("empty", last_error="out of shared memory")]) + with pytest.raises(RuntimeError, match="out of shared memory"): + client._ensure_graph_cache_ready() + client._call_graph_cache_function.assert_not_called() + + def test_timeout_reports_last_state(self, monkeypatch: pytest.MonkeyPatch): + client = make_client([cache_detail("building")], graph_cache_timeout=1) + monotonic = MagicMock(side_effect=[0.0, 2.0]) + monkeypatch.setattr("vectordb_bench.backend.clients.polardb_pg.polardb_pg.time.monotonic", monotonic) + + with pytest.raises(TimeoutError, match="after 1s.*status=building"): + client._ensure_graph_cache_ready() + + +def test_polardb_pg_cli_help(): + result = CliRunner().invoke(PolarDBPgHNSWCommand, ["--help"]) + + assert result.exit_code == 0, result.output + assert PolarDBPgHNSWCommand.name == "polardbpghnsw" + assert "--graph-cache / --skip-graph-cache" in result.output + assert "--graph-cache-poll-interval" not in result.output + assert "--graph-cache-timeout INTEGER" in result.output + assert "--quantization [pq|sq4|sq8|rabitq]" in result.output + assert "--quantization-nbits [1|4|8]" in result.output + + +def test_polardb_pg_cli_dry_run_builds_task( + monkeypatch: pytest.MonkeyPatch, + caplog: pytest.LogCaptureFixture, +): + benchmark_run = MagicMock() + monkeypatch.setattr("vectordb_bench.cli.cli.benchmark_runner.run", benchmark_run) + caplog.set_level("INFO", logger="vectordb_bench.cli.cli") + + result = CliRunner().invoke( + PolarDBPgHNSWCommand, + [ + "--case-type", + "Performance1024D1M", + "--db-label", + "polardb-pg-dry-run", + "--user-name", + "postgres", + "--password", + "secret", + "--host", + "localhost", + "--port", + "5432", + "--db-name", + "vec", + "--m", + "16", + "--ef-construction", + "256", + "--ef-search", + "200", + "--maintenance-work-mem", + "128GB", + "--max-parallel-workers", + "64", + "--quantization-type", + "none", + "--table-quantization-type", + "none", + "--skip-reranking", + "--quantized-fetch-limit", + "400", + "--iterative-scan", + "off", + "--skip-create-index-before-load", + "--create-index-after-load", + "--graph-cache", + "--graph-cache-timeout", + "120", + "--quantization", + "rabitq", + "--train-samples", + "5000", + "--quantization-nbits", + "8", + "--dry-run", + ], + ) + + assert result.exit_code == 0, result.output + benchmark_run.assert_not_called() + assert "graph_cache_timeout=120" in caplog.text + assert "quantization='rabitq'" in caplog.text + assert "secret" not in caplog.text + + +class TestPolarDBPgFrontend: + @staticmethod + def input_map(case_label: CaseLabel): + return {config.label: config for config in get_case_config_inputs(DB.PolarDBPG, case_label)} + + def test_uses_hnsw_only_and_exposes_polardb_options(self): + load_inputs = self.input_map(CaseLabel.Load) + performance_inputs = self.input_map(CaseLabel.Performance) + + assert load_inputs[CaseConfigParamType.IndexType].inputConfig["options"] == [IndexType.HNSW.value] + assert CaseConfigParamType.hnsw_quantization in load_inputs + assert CaseConfigParamType.quantization_nbits in load_inputs + assert CaseConfigParamType.graph_cache in load_inputs + assert CaseConfigParamType.graph_cache_timeout in load_inputs + assert CaseConfigParamType.post_load_index in load_inputs + assert CaseConfigParamType.ef_search in performance_inputs + assert CaseConfigParamType.graph_cache_timeout in performance_inputs + assert CaseConfigParamType.iterative_scan in performance_inputs + + def test_quantization_parameter_has_a_distinct_enum_value(self): + assert CaseConfigParamType.hnsw_quantization.value == "hnsw_quantization" + assert CaseConfigParamType.hnsw_quantization is not CaseConfigParamType.mongodb_quantization_type + + def test_quantizer_fields_are_conditionally_displayed(self): + inputs = self.input_map(CaseLabel.Performance) + config = { + CaseConfigParamType.IndexType: IndexType.HNSW.value, + CaseConfigParamType.hnsw_quantization: "rabitq", + CaseConfigParamType.graph_cache: True, + } + + assert inputs[CaseConfigParamType.train_samples].isDisplayed(config) + assert inputs[CaseConfigParamType.quantization_nbits].isDisplayed(config) + assert not inputs[CaseConfigParamType.pq_m].isDisplayed(config) + + config[CaseConfigParamType.hnsw_quantization] = "pq" + assert inputs[CaseConfigParamType.pq_m].isDisplayed(config) + assert not inputs[CaseConfigParamType.quantization_nbits].isDisplayed(config) + + def test_iterative_scan_is_hidden_with_graph_cache(self): + inputs = self.input_map(CaseLabel.Performance) + config = {CaseConfigParamType.graph_cache: True} + + assert inputs[CaseConfigParamType.graph_cache_timeout].isDisplayed(config) + assert not inputs[CaseConfigParamType.iterative_scan].isDisplayed(config) + + config[CaseConfigParamType.graph_cache] = False + assert not inputs[CaseConfigParamType.graph_cache_timeout].isDisplayed(config) + assert inputs[CaseConfigParamType.iterative_scan].isDisplayed(config) diff --git a/vectordb_bench/backend/clients/__init__.py b/vectordb_bench/backend/clients/__init__.py index dbd83bb66..cfc4bf15c 100644 --- a/vectordb_bench/backend/clients/__init__.py +++ b/vectordb_bench/backend/clients/__init__.py @@ -30,6 +30,7 @@ class DB(Enum): QdrantLocal = "QdrantLocal" WeaviateCloud = "WeaviateCloud" PgVector = "PgVector" + PolarDBPG = "PolarDBPG" PgVectoRS = "PgVectoRS" PgVectorScale = "PgVectorScale" PgDiskANN = "PgDiskANN" @@ -110,6 +111,11 @@ def init_cls(self) -> type[VectorDB]: # noqa: PLR0911, PLR0912, C901, PLR0915 return PgVector + if self == DB.PolarDBPG: + from .polardb_pg.polardb_pg import PolarDBPgHNSW + + return PolarDBPgHNSW + if self == DB.PgVectoRS: from .pgvecto_rs.pgvecto_rs import PgVectoRS @@ -333,6 +339,11 @@ def config_cls(self) -> type[DBConfig]: # noqa: PLR0911, PLR0912, C901, PLR0915 return PgVectorConfig + if self == DB.PolarDBPG: + from .polardb_pg.config import PolarDBPgConfig + + return PolarDBPgConfig + if self == DB.PgVectoRS: from .pgvecto_rs.config import PgVectoRSConfig @@ -560,6 +571,11 @@ def case_config_cls( # noqa: C901, PLR0911, PLR0912, PLR0915 return _pgvector_case_config.get(index_type) + if self == DB.PolarDBPG: + from .polardb_pg.config import _polardb_pg_case_config + + return _polardb_pg_case_config.get(index_type) + if self == DB.PgVectoRS: from .pgvecto_rs.config import _pgvecto_rs_case_config diff --git a/vectordb_bench/backend/clients/pgvector/pgvector.py b/vectordb_bench/backend/clients/pgvector/pgvector.py index 13f471afe..c60b5d69b 100644 --- a/vectordb_bench/backend/clients/pgvector/pgvector.py +++ b/vectordb_bench/backend/clients/pgvector/pgvector.py @@ -19,9 +19,17 @@ log = logging.getLogger(__name__) +def _redact_connect_config(connect_config: dict[str, Any]) -> dict[str, Any]: + redacted = dict(connect_config) + if "password" in redacted: + redacted["password"] = "**********" # noqa: S105 + return redacted + + class PgVector(VectorDB): """Use psycopg instructions""" + name = "PgVector" thread_safe: bool = False supported_filter_types: list[FilterOp] = [ FilterOp.NonFilter, @@ -43,7 +51,6 @@ def __init__( with_scalar_labels: bool = False, **kwargs, ): - self.name = "PgVector" self.case_config = db_case_config self.table_name = db_config["table_name"] self.connect_config = db_config["connect_config"] @@ -62,7 +69,8 @@ def __init__( self.cursor.execute("CREATE EXTENSION IF NOT EXISTS vector") self.conn.commit() - log.info(f"{self.name} config values: {self.connect_config}\n{self.case_config}") + redacted_connect_config = _redact_connect_config(self.connect_config) + log.info("%s config values: %s\n%s", self.name, redacted_connect_config, self.case_config) if not any( ( self.case_config.create_index_before_load, @@ -71,7 +79,7 @@ def __init__( ): msg = ( f"{self.name} config must create an index using create_index_before_load or create_index_after_load" - f"{self.name} config values: {self.connect_config}\n{self.case_config}" + f"{self.name} config values: {redacted_connect_config}\n{self.case_config}" ) log.error(msg) raise RuntimeError(msg) diff --git a/vectordb_bench/backend/clients/polardb_pg/__init__.py b/vectordb_bench/backend/clients/polardb_pg/__init__.py new file mode 100644 index 000000000..10a4ff541 --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/__init__.py @@ -0,0 +1 @@ +"""PolarDB for PostgreSQL vector benchmark client.""" diff --git a/vectordb_bench/backend/clients/polardb_pg/cli.py b/vectordb_bench/backend/clients/polardb_pg/cli.py new file mode 100644 index 000000000..8fd737074 --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/cli.py @@ -0,0 +1,139 @@ +from typing import Annotated, TypedDict, Unpack + +import click +from pydantic import SecretStr + +from vectordb_bench.backend.clients import DB + +from ....cli.cli import ( + HNSWFlavor1, + cli, + click_parameter_decorators_from_typed_dict, + get_custom_case_config, + run, +) +from ..pgvector.cli import PgVectorTypedDict +from .config import DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS + + +def parse_quantization_nbits(ctx: object, param: object, value: str | None) -> int | None: # noqa: ARG001 + return int(value) if value is not None else None + + +class PolarDBPgHNSWOptions(TypedDict): + iterative_scan: Annotated[ + str, + click.option( + "--iterative-scan", + type=click.Choice(["off", "strict_order", "relaxed_order"]), + default="off", + show_default=True, + help="HNSW iterative scan mode; Graph Cache requires off", + ), + ] + create_index_before_load: Annotated[ + bool, + click.option( + "--create-index-before-load/--skip-create-index-before-load", + default=False, + show_default=True, + help="Create the HNSW index before loading data", + ), + ] + create_index_after_load: Annotated[ + bool, + click.option( + "--create-index-after-load/--skip-create-index-after-load", + default=True, + show_default=True, + help="Create the HNSW index after loading data", + ), + ] + graph_cache: Annotated[ + bool, + click.option( + "--graph-cache/--skip-graph-cache", + default=True, + show_default=True, + help="Build Graph Cache and wait until it is usable before search", + ), + ] + graph_cache_timeout: Annotated[ + int, + click.option( + "--graph-cache-timeout", + type=click.IntRange(min=1), + default=DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS, + show_default=True, + help="Seconds to wait for Graph Cache to become usable", + ), + ] + quantization: Annotated[ + str | None, + click.option( + "--quantization", + type=click.Choice(["pq", "sq4", "sq8", "rabitq"]), + help="PolarDB HNSW internal quantization method", + ), + ] + pq_m: Annotated[ + int | None, + click.option("--pq-m", type=int, help="Number of PQ sub-quantizers"), + ] + train_samples: Annotated[ + int | None, + click.option("--train-samples", type=int, help="Number of quantizer training samples"), + ] + quantization_nbits: Annotated[ + int | None, + click.option( + "--quantization-nbits", + type=click.Choice(["1", "4", "8"]), + callback=parse_quantization_nbits, + help="RaBitQ bits per dimension", + ), + ] + + +class PolarDBPgHNSWTypedDict(PgVectorTypedDict, HNSWFlavor1, PolarDBPgHNSWOptions): ... + + +@cli.command() +@click_parameter_decorators_from_typed_dict(PolarDBPgHNSWTypedDict) +def PolarDBPgHNSW(**parameters: Unpack[PolarDBPgHNSWTypedDict]): + from .config import PolarDBPgConfig, PolarDBPgHNSWConfig + + parameters["custom_case"] = get_custom_case_config(parameters) + run( + db=DB.PolarDBPG, + db_config=PolarDBPgConfig( + db_label=parameters["db_label"], + user_name=SecretStr(parameters["user_name"]), + password=SecretStr(parameters["password"]), + host=parameters["host"], + port=parameters["port"], + db_name=parameters["db_name"], + ), + db_case_config=PolarDBPgHNSWConfig( + m=parameters["m"], + ef_construction=parameters["ef_construction"], + ef_search=parameters["ef_search"], + maintenance_work_mem=parameters["maintenance_work_mem"], + max_parallel_workers=parameters["max_parallel_workers"], + quantization_type=parameters["quantization_type"], + table_quantization_type=parameters["table_quantization_type"], + reranking=parameters["reranking"], + reranking_metric=parameters["reranking_metric"], + quantized_fetch_limit=parameters["quantized_fetch_limit"], + iterative_scan=parameters["iterative_scan"], + create_index_before_load=parameters["create_index_before_load"], + create_index_after_load=parameters["create_index_after_load"], + graph_cache=parameters["graph_cache"], + graph_cache_timeout=parameters["graph_cache_timeout"], + quantization=parameters["quantization"], + pq_m=parameters["pq_m"], + train_samples=parameters["train_samples"], + quantization_nbits=parameters["quantization_nbits"], + ), + **parameters, + ) diff --git a/vectordb_bench/backend/clients/polardb_pg/config.py b/vectordb_bench/backend/clients/polardb_pg/config.py new file mode 100644 index 000000000..f1e12bc35 --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/config.py @@ -0,0 +1,71 @@ +from typing import Literal + +from pydantic import AliasChoices, Field, PositiveInt, model_validator + +from ..api import IndexType +from ..pgvector.config import PgVectorConfig, PgVectorHNSWConfig, PgVectorIndexParam + +DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS = 3600 + + +class PolarDBPgConfig(PgVectorConfig): + """Connection configuration for PolarDB for PostgreSQL.""" + + +class PolarDBPgHNSWConfig(PgVectorHNSWConfig): + """PolarDB HNSW-specific benchmark options.""" + + post_load_index: bool | None = None + iterative_scan: str = "off" + graph_cache: bool = True + graph_cache_timeout: PositiveInt = DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS + quantization: Literal["pq", "sq4", "sq8", "rabitq"] | None = Field( + default=None, + validation_alias=AliasChoices("quantization", "hnsw_quantization"), + ) + pq_m: int | None = None + train_samples: int | None = None + quantization_nbits: Literal[1, 4, 8] | None = None + + @model_validator(mode="after") + def validate_polardb_options(self) -> "PolarDBPgHNSWConfig": + if self.post_load_index is not None: + self.create_index_before_load = not self.post_load_index + self.create_index_after_load = self.post_load_index + if self.iterative_scan not in {"off", "strict_order", "relaxed_order"}: + msg = "iterative_scan must be one of: off, strict_order, relaxed_order" + raise ValueError(msg) + if self.graph_cache and self.iterative_scan != "off": + msg = "Graph Cache requires iterative_scan=off" + raise ValueError(msg) + if self.pq_m is not None and self.quantization != "pq": + msg = "pq_m is only valid with quantization=pq" + raise ValueError(msg) + if self.quantization_nbits is not None and self.quantization != "rabitq": + msg = "quantization_nbits is only valid with quantization=rabitq" + raise ValueError(msg) + if self.train_samples is not None and self.quantization is None: + msg = "train_samples requires a quantization method" + raise ValueError(msg) + return self + + def index_param(self) -> PgVectorIndexParam: + index_param = super().index_param() + polar_options = { + "quantization": self.quantization, + "pq_m": self.pq_m, + "train_samples": self.train_samples, + "quantization_nbits": self.quantization_nbits, + "cache": "on" if self.graph_cache else None, + } + index_param["index_creation_with_options"] = [ + *index_param["index_creation_with_options"], + *self._optionally_build_with_options(polar_options), + ] + return index_param + + +_polardb_pg_case_config = { + IndexType.HNSW: PolarDBPgHNSWConfig, + IndexType.ES_HNSW: PolarDBPgHNSWConfig, +} diff --git a/vectordb_bench/backend/clients/polardb_pg/polardb_pg.py b/vectordb_bench/backend/clients/polardb_pg/polardb_pg.py new file mode 100644 index 000000000..fe50261bd --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/polardb_pg.py @@ -0,0 +1,176 @@ +"""PolarDB for PostgreSQL HNSW client.""" + +from __future__ import annotations + +import logging +import re +import time +from dataclasses import dataclass +from typing import TYPE_CHECKING + +from psycopg import sql + +from ..pgvector.pgvector import PgVector + +if TYPE_CHECKING: + from .config import PolarDBPgHNSWConfig + +log = logging.getLogger(__name__) + +_GRAPH_CACHE_POLL_INTERVAL_SECONDS = 1.0 + + +@dataclass(frozen=True) +class HNSWGraphCacheDetail: + requested: str + status: str + usable: bool + last_error: str | None + raw: str + + @staticmethod + def _field(detail: str, name: str) -> str: + match = re.search(rf"(?:^|\s){re.escape(name)}=([^\s]+)", detail) + if match is None: + msg = f"Graph Cache detail is missing {name}: {detail}" + raise RuntimeError(msg) + return match.group(1) + + @classmethod + def parse(cls, detail: str) -> HNSWGraphCacheDetail: + requested = cls._field(detail, "requested") + status = cls._field(detail, "status") + usable_value = cls._field(detail, "usable") + if usable_value not in {"yes", "no"}: + msg = f"Graph Cache detail has invalid usable state: {detail}" + raise RuntimeError(msg) + error_match = re.search(r"(?:^|\s)last_error=(.*)$", detail) + last_error = error_match.group(1) if error_match else None + if last_error == "none": + last_error = None + return cls( + requested=requested, + status=status, + usable=usable_value == "yes", + last_error=last_error, + raw=detail, + ) + + +class PolarDBPgHNSW(PgVector): + """PgVector-compatible PolarDB HNSW client.""" + + name = "PolarDBPG" + case_config: PolarDBPgHNSWConfig + + def __init__(self, *args, drop_old: bool = False, **kwargs): + super().__init__(*args, drop_old=drop_old, **kwargs) + + # Search-only benchmarks skip optimize(), so validate and prepare the + # existing index before any search workers can start. + if not drop_old and self.case_config.graph_cache: + with self.init(): + self._ensure_graph_cache_ready() + + def _post_insert(self): + super()._post_insert() + if self.case_config.graph_cache: + self._ensure_graph_cache_ready() + + def _qualified_index_name(self) -> str: + return f"public.{self._index_name}" + + def _graph_cache_detail(self) -> HNSWGraphCacheDetail: + assert self.conn is not None, "Connection is not initialized" + assert self.cursor is not None, "Cursor is not initialized" + + row = self.cursor.execute( + "SELECT hnsw_cache_detail(%s::regclass)", + (self._qualified_index_name(),), + ).fetchone() + self.conn.commit() + if row is None or row[0] is None: + msg = f"Unable to read Graph Cache state for {self._qualified_index_name()}" + raise RuntimeError(msg) + return HNSWGraphCacheDetail.parse(row[0]) + + def _call_graph_cache_function(self, function_name: str) -> bool: + assert self.conn is not None, "Connection is not initialized" + assert self.cursor is not None, "Cursor is not initialized" + + statement = sql.SQL("SELECT {}(%s::regclass)").format(sql.Identifier(function_name)) + row = self.cursor.execute(statement, (self._qualified_index_name(),)).fetchone() + self.conn.commit() + return bool(row and row[0]) + + def _check_graph_cache_role(self): + assert self.conn is not None, "Connection is not initialized" + assert self.cursor is not None, "Cursor is not initialized" + + row = self.cursor.execute("SELECT pg_is_in_recovery()").fetchone() + self.conn.commit() + if row and row[0]: + msg = "PolarDB HNSW Graph Cache benchmarks must run on the primary node" + raise RuntimeError(msg) + + def _ensure_graph_cache_ready(self): + self._check_graph_cache_role() + timeout = self.case_config.graph_cache_timeout + deadline = time.monotonic() + timeout + last_logged_state: tuple[str, bool] | None = None + last_progress_log = 0.0 + release_requested = False + + while True: + detail = self._graph_cache_detail() + if detail.requested != "on": + msg = ( + f"HNSW index {self._qualified_index_name()} was not created with cache=on; " + "enable that reloption, rebuild the index, or run with --skip-graph-cache" + ) + raise RuntimeError(msg) + if detail.status == "not_preloaded": + msg = ( + "HNSW Graph Cache is not preloaded; add vector to " + "shared_preload_libraries and restart PostgreSQL" + ) + raise RuntimeError(msg) + if detail.status == "ready" and detail.usable: + log.info("HNSW Graph Cache is ready for %s", self._qualified_index_name()) + return + now = time.monotonic() + state = (detail.status, detail.usable) + if state != last_logged_state or now - last_progress_log >= 60: + log.info( + "Waiting for HNSW Graph Cache: index=%s status=%s usable=%s", + self._qualified_index_name(), + detail.status, + "yes" if detail.usable else "no", + ) + last_logged_state = state + last_progress_log = now + + if detail.status == "ready": + if not release_requested: + log.info("Rebuilding stale HNSW Graph Cache for %s", self._qualified_index_name()) + self._call_graph_cache_function("hnsw_release_cache") + release_requested = True + elif detail.status == "empty": + if detail.last_error: + msg = f"HNSW Graph Cache build failed for {self._qualified_index_name()}: {detail.last_error}" + raise RuntimeError(msg) + accepted = self._call_graph_cache_function("hnsw_schedule_cache_rebuild") + if accepted: + release_requested = False + log.info("Scheduled HNSW Graph Cache build for %s", self._qualified_index_name()) + elif detail.status not in {"building", "draining"}: + msg = f"Unsupported HNSW Graph Cache state for {self._qualified_index_name()}: {detail.raw}" + raise RuntimeError(msg) + + if now >= deadline: + msg = ( + f"Timed out after {timeout:g}s waiting for " + f"HNSW Graph Cache on {self._qualified_index_name()}: {detail.raw}" + ) + raise TimeoutError(msg) + time.sleep(min(_GRAPH_CACHE_POLL_INTERVAL_SECONDS, max(deadline - now, 0))) diff --git a/vectordb_bench/cli/vectordbbench.py b/vectordb_bench/cli/vectordbbench.py index 1bbc462ef..e4d274a52 100644 --- a/vectordb_bench/cli/vectordbbench.py +++ b/vectordb_bench/cli/vectordbbench.py @@ -41,6 +41,7 @@ PolarDBHNSWPQ, PolarDBHNSWSQ, ) +from ..backend.clients.polardb_pg.cli import PolarDBPgHNSW from ..backend.clients.qdrant_cloud.cli import QdrantCloud from ..backend.clients.qdrant_local.cli import QdrantLocal from ..backend.clients.redis.cli import Redis @@ -61,6 +62,7 @@ cli.add_command(AdbpgNova) cli.add_command(PgVectorHNSW) +cli.add_command(PolarDBPgHNSW) cli.add_command(PgVectoRSHNSW) cli.add_command(PgVectoRSIVFFlat) cli.add_command(Redis) diff --git a/vectordb_bench/frontend/config/dbCaseConfigs.py b/vectordb_bench/frontend/config/dbCaseConfigs.py index 678b53288..f23ff3614 100644 --- a/vectordb_bench/frontend/config/dbCaseConfigs.py +++ b/vectordb_bench/frontend/config/dbCaseConfigs.py @@ -6,6 +6,7 @@ from vectordb_bench.backend.cases import CaseLabel, CaseType, PerformanceCase from vectordb_bench.backend.clients import DB from vectordb_bench.backend.clients.api import IndexType, MetricType, SQType +from vectordb_bench.backend.clients.polardb_pg.config import DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS from vectordb_bench.backend.dataset import ( LAION_INT_FILTER_SEARCH_WIDTHS, DatasetManager, @@ -1594,6 +1595,96 @@ class CaseConfigInput(BaseModel): ) +# PolarDB for PostgreSQL HNSW configs +CaseConfigParamInput_IndexType_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.IndexType, + inputHelp="PolarDB for PostgreSQL currently supports the HNSW index in VectorDBBench", + inputType=InputType.Option, + inputConfig={"options": [IndexType.HNSW.value]}, +) + +CaseConfigParamInput_HnswQuantization_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.hnsw_quantization, + displayLabel="Internal quantization", + inputHelp="PolarDB HNSW internal quantizer; None keeps the standard pgvector code format", + inputType=InputType.Option, + inputConfig={"options": [None, "pq", "sq4", "sq8", "rabitq"]}, +) + +CaseConfigParamInput_PostLoadIndex_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.post_load_index, + displayLabel="Create index after load", + inputHelp="Disable to build the HNSW index before loading data", + inputType=InputType.Bool, + inputConfig={"value": True}, +) + +CaseConfigParamInput_PQM_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.pq_m, + displayLabel="PQ sub-quantizers", + inputHelp="Number of sub-quantizers used by PQ", + inputType=InputType.Number, + inputConfig={ + "min": 2, + "max": 4096, + "value": 32, + }, + isDisplayed=lambda config: config.get(CaseConfigParamType.hnsw_quantization, None) == "pq", +) + +CaseConfigParamInput_TrainSamples_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.train_samples, + displayLabel="Quantizer train samples", + inputHelp="Number of sampled rows used to train the internal quantizer", + inputType=InputType.Number, + inputConfig={ + "min": 100, + "max": MAX_STREAMLIT_INT, + "value": 1000, + }, + isDisplayed=lambda config: config.get(CaseConfigParamType.hnsw_quantization, None) is not None, +) + +CaseConfigParamInput_QuantizationNbits_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.quantization_nbits, + displayLabel="RaBitQ bits per dimension", + inputHelp="RaBitQ code width", + inputType=InputType.Option, + inputConfig={"options": [1, 4, 8]}, + isDisplayed=lambda config: config.get(CaseConfigParamType.hnsw_quantization, None) == "rabitq", +) + +CaseConfigParamInput_GraphCache_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.graph_cache, + displayLabel="Enable Graph Cache", + inputHelp="Build Graph Cache and wait until it is usable before search", + inputType=InputType.Bool, + inputConfig={"value": True}, +) + +CaseConfigParamInput_GraphCacheTimeout_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.graph_cache_timeout, + displayLabel="Graph Cache timeout (seconds)", + inputHelp="Maximum time to wait for Graph Cache to become usable", + inputType=InputType.Number, + inputConfig={ + "min": 1, + "max": MAX_STREAMLIT_INT, + "value": DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS, + }, + isDisplayed=lambda config: config.get(CaseConfigParamType.graph_cache, True), +) + +CaseConfigParamInput_IterativeScan_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.iterative_scan, + displayLabel="Iterative scan", + inputHelp="Graph Cache requires iterative scan to remain off", + inputType=InputType.Option, + inputConfig={"options": ["off", "strict_order", "relaxed_order"]}, + isDisplayed=lambda config: not config.get(CaseConfigParamType.graph_cache, True), +) + + CaseConfigParamInput_IndexType_AlloyDB = CaseConfigInput( label=CaseConfigParamType.IndexType, inputHelp="Select Index Type", @@ -2554,6 +2645,45 @@ class CaseConfigInput(BaseModel): CaseConfigParamInput_quantized_fetch_limit_PgVector, ] +PolarDBPgLoadingConfig = [ + CaseConfigParamInput_IndexType_PolarDBPG, + CaseConfigParamInput_PostLoadIndex_PolarDBPG, + CaseConfigParamInput_m, + CaseConfigParamInput_EFConstruction_PgVector, + CaseConfigParamInput_QuantizationType_PgVector, + CaseConfigParamInput_TableQuantizationType_PgVector, + CaseConfigParamInput_HnswQuantization_PolarDBPG, + CaseConfigParamInput_PQM_PolarDBPG, + CaseConfigParamInput_TrainSamples_PolarDBPG, + CaseConfigParamInput_QuantizationNbits_PolarDBPG, + CaseConfigParamInput_maintenance_work_mem_PgVector, + CaseConfigParamInput_max_parallel_workers_PgVector, + CaseConfigParamInput_GraphCache_PolarDBPG, + CaseConfigParamInput_GraphCacheTimeout_PolarDBPG, +] + +PolarDBPgPerformanceConfig = [ + CaseConfigParamInput_IndexType_PolarDBPG, + CaseConfigParamInput_PostLoadIndex_PolarDBPG, + CaseConfigParamInput_m, + CaseConfigParamInput_EFConstruction_PgVector, + CaseConfigParamInput_EFSearch_PgVector, + CaseConfigParamInput_QuantizationType_PgVector, + CaseConfigParamInput_TableQuantizationType_PgVector, + CaseConfigParamInput_HnswQuantization_PolarDBPG, + CaseConfigParamInput_PQM_PolarDBPG, + CaseConfigParamInput_TrainSamples_PolarDBPG, + CaseConfigParamInput_QuantizationNbits_PolarDBPG, + CaseConfigParamInput_maintenance_work_mem_PgVector, + CaseConfigParamInput_max_parallel_workers_PgVector, + CaseConfigParamInput_reranking_PgVector, + CaseConfigParamInput_reranking_metric_PgVector, + CaseConfigParamInput_quantized_fetch_limit_PgVector, + CaseConfigParamInput_GraphCache_PolarDBPG, + CaseConfigParamInput_GraphCacheTimeout_PolarDBPG, + CaseConfigParamInput_IterativeScan_PolarDBPG, +] + PgVectoRSLoadingConfig = [ CaseConfigParamInput_IndexType_PgVectoRS, CaseConfigParamInput_m, @@ -3444,6 +3574,10 @@ class FilterType(Enum): CaseLabel.Load: PgVectorLoadingConfig, CaseLabel.Performance: PgVectorPerformanceConfig, }, + DB.PolarDBPG: { + CaseLabel.Load: PolarDBPgLoadingConfig, + CaseLabel.Performance: PolarDBPgPerformanceConfig, + }, DB.PgVectoRS: { CaseLabel.Load: PgVectoRSLoadingConfig, CaseLabel.Performance: PgVectoRSPerformanceConfig, diff --git a/vectordb_bench/frontend/config/styles.py b/vectordb_bench/frontend/config/styles.py index 4b162e9ed..eb9186cfc 100644 --- a/vectordb_bench/frontend/config/styles.py +++ b/vectordb_bench/frontend/config/styles.py @@ -48,6 +48,7 @@ def getPatternShape(i): DB.QdrantLocal: "https://assets.zilliz.com/qdrant_b691674fcd.png", DB.WeaviateCloud: "https://assets.zilliz.com/weaviate_4f6f171ebe.png", DB.PgVector: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", + DB.PolarDBPG: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", DB.PgVectoRS: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", DB.PgVectorScale: "data:image/jpeg;base64,/9j/4AAQSkZJRgABAQAAAQABAAD/2wCEAAkGBxAQEhUQEBAWEBAQEBUXEBIVFxgRERASGhUYFhkXGBgaHiggHR0nHRYVLTEhJSkrLjouFx8/OTMsNyg5MS0BCgoKDg0OGBAQGC0eHx8wLS0tLS03KzEtNysrLS0tLSs3Ky03LSstLS0tKyswNi0rNy0tMC4tLSstLS0tLS0tLf/AABEIAMgAyAMBIgACEQEDEQH/xAAcAAEAAgIDAQAAAAAAAAAAAAAABwgBBQMEBgL/xABFEAABAwICBgUJAg0EAwAAAAABAAIDBBEFIQYHEjFBUSIyYXGBExQjQlJikaGxCMEVJDM0Q3J0gpKTorLCFlVj0VNz0v/EABoBAQACAwEAAAAAAAAAAAAAAAACBQEDBgT/xAAqEQACAgIBAwIGAgMAAAAAAAAAAQIDBBExBRIhE0EiMlFSYfCBkTShsf/aAAwDAQACEQMRAD8Ag1ERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAWVutHtF62vds0sDpBexfbZjb3vOQUn4DqN3Oraq3OOAf5u/6Wi3Jrq+ZjRCqyrR4XqzwinGVI2V3tSkyk+By+S9FS4PTRC0VPFGPcja36BeKXVa18sdku0p3snksWVz/ACTfZHwXXqMLp5BaSnikHJzGu+oUF1aP2/7HaU4WFajE9XGE1A6VGxh9qK8JH8OS8Pjuo1hBdRVRaeEcw2m/xt3fAr0V9Spl4fgx2kIIvQaR6G1+Hk+cwOay+UrenEf3hl8V59e6MlJbT2YMIiLICIiAIiIAiIgCIiAIiIAiytjgWDz1szaenZtyPPg0cXOPADmjaS2wdWio5J3tiiY6SR5s1jRtOcewKa9B9TjGBs2J9N+8U7T0G/8AscOt3DLvXstAtBKfC48gJapw9LORn+qz2W/XivXqjyuot/DX4X1JKJwUtNHE0RxMbGxos1jQGtaOwBc6Ljlka0FziGtaCXOOQaBmSTyVX5kyR9oo7w3W3QTVjqU3jivsw1Dj0JHdo9Ucj8bKQwVsspnX8y0Nn0iItICIiA45YmvBa5oc1wsWkXaR2hRdptqfgnDpsPtTzb/JH8g89nFh+XcpUWVvpyJ1PcWOSnOK4XPSyOgqI3RSs3tcLeI5jtXSVsNM9EKbE4vJzN2ZGj0UwHTjP3jsVadKtG6jDpzT1DbHex46krODmldBi5cb19GQa0aREReswEREAREQBERAEREB2KKkfM9sUTS+SRwaxo3uccgFZzVzoVHhcFiA6qlAM8n+DfdHzXi9Reh4a04nM3pOu2lBHVbudJ45gdx5qY1SdRytv048LklFGURFUEgow18Y66CjZTMdsuq3u27bzEyxcPElnwKk9Qr9oqld+KTeoPKsJ5OOy4fEA/BezAinfFMw+CFlKWrPWg+j2aWuJfS7o5Os+Abre8zs3j5KLUXRW1Rtj2yRAuBhePUdV+b1MU3Yx7XO+G9bJUva4g3GRG48l6LCtOsUpreSrZbDc158q23c+4VVZ0r7Jf2S7i1yKAcK131rMqininHEtvC8+OY+S9vgeuPDqhzY5WSUznkDafsuiBPN4N7dpC8dmBdD22Z2iSEXy0g5jMHcea+l42tGTC89ptorDidOYJOi8XMMts4n8+7dcL0KKUJuElKPIKd4zhctJM+nnbsSxOs4fQjsIzXRVg9dmiAqqfz6Fvp6ZvpAN8kG8+Lb37rqvi6jGvV0FL3INaMIiLeYCIiAIiIAtxongrq6rhpW39LIA4j1Wb3u8BdahTL9nvBbunrXDqAQxHtPSf8ALY/iK05FvpVuRlEz0dKyGNsUbQ2ONgaxo4NAsFzrCyuVb29smERFEBaPTHR6LEaV9NLlfpRvGZjkG5w+fgSvJ6wNaUWHytp6djaiZrh5xc2bG32Lj1/ovX6P6QU+IU4qKd+00izmnrRvtm1w5r1Km2pRs1oFWMfwWaimdBMLOaciOq9vBwPJaxTTrKoWTAhwzabtdxaVDM0ZaS07wV0OPd6kE/cgzjREW8wEREBLeqjWV5vs0Nc/0GQgmcfyPJjj7Hbw7t07tcDmMwRkeapcpa1U6y/Ny2hrn3gOUMzj+R4Bjvc7eHduqs7B7tzgvJJMnhFhrgcxmCMjzWQqNokfMjA4EOAIIsQcwQqq6w9HvwfXSwAeiJ24O2J2YHhmP3Va1RH9oLBQ+CGsaOlC/wAm8/8AG/MX7nD+tWPTruy3t9mYkQOiIugIBERAEREBlWf1Q4cKfC4PamDpXfvnL+kNVYArg6OQCOkp4x6lNE34MAVZ1SWq0vqyUTZIiKhJGFptL4qt9JM2hcGVRZ6Mn5hp4Otex5rQaR6zqGiqmUj7yEutPIw3bT8BfmeYG75L2kEzZGh7HB7HAFrmm7XA5gg8lu9OdTjOSBTisikY9zZQ5sgcRIHdYO43vne63Wh2ldRhk3lYjdjspoj1JWfceRU260NXbMQaammAZWsbnwbUAeq73uR8D2V2qIHxucx7Sx7HEOaRZzXDIghdDRdDJh/1EGtEuY9jMNZD5eF12uGYPWY7i13aolruu7vXLQYg+EnZPRdk5vBwXXqXhzi4biVsqq9PwuAcSIi3GAiIgCIiAlvVRrLNPs0Nc+8BygmdmYeAa73Pp3bp3a4EXGYIyPNUuU56h9JqicSUMrttkEYfC45uY2+zsd2eXJVGfhrTsj/JJP2JfXntYWHipw6qiIufIOc39ZnpG/NoXoV8TxB7XMO5zS09xFlUVS7Zp/Qkyl6Lmqo9h7m+y4j4Gy4V1yNYREQBERAZCuXRfk2W3bDbfAKmgVwtH5xJSwSD16eJ3xYCqnqq+GP8komxXjtaeKVtLQvkomXde0sgzdBFY3e0ffw3r0Nbi8EEsMEsgZJUl4gB9ctsSL7r9Id67zm3FjmCMxwKqa36coyktokUwe8uJJNyTck5klSLqt1jOw9wpakl1E85He6nceI93mPEdux1r6tvNtqtom/i5N5oR+hPtN9z6d26J10i9PJr/BDgudBM2Roexwex7Q5rmm7XA53BUf60NXTcRaammAZWsbmNzaho9V3vcj4d0c6rtYzqBwpqkl1G45He6nJ4j3eY8e+w0EzXtD2ODmOALXNN2uBzBBVJZXZiWbXBLkptUwPjc5j2lj2OIc1ws5pG8ELiVjtaGrtuINNTTAMrWNz4NqGgdU+9yPh3V2nhdG4se0te1xDmkWLXDIg9qvMbJjfHa5ItaOFERbzAREQBERAZUxfZ4oJPK1NRb0QjbGDzeXB1h3AfMKJaCkfPIyGNpdJI8NYBxcTYK2Gh+Asw+kipWWuxt5HD15Dm53x+QC8HUblCrt92ZibtYWVxyyBrS47mgk9wzXPR5RMp7jP5xNbd5eS38ZXRXNWS7b3O9p7j8TdcK69cGsIiLICIiAyrR6psRE+F05vcxNMTuwsJA/p2fiquKa/s9YzlUUTjuImiH9D/APD5rwdRr7qdr2Mrk+vtFOI8xINiHVFjxB9Cu/qo1lCpDaKtfaoAtDKchP7rj7f1710ftFxOLaJ1jstdUBzuAJEVh/S74KE2PINwbEG4O4gqFFEbsWMX++TLemXDxcXjIOYIz7VW7T7RlsEjpYBaNxu5g9Q9nYve6vtZPnUYoq11qgC0MpyE43bLvf8Ar3ro6ZDetGNCzHscWH5IcUk6rNYr6Bwpakl1G92R3mnceI93mPHvjysaA8gbrrhCtbK42x7ZIiXRa4EXGYIyPNV7184M2CtZUMGyKuK77cZGHZcfEFinbAPzWD9ni/sCiL7RvWov1Z/rEqPAbjkdq/JN8EMLKwt/SaHYlMxskVDO+N4u1wjdsuHMdiv3JR5ZA0CL0f8AoTFv9vn/AJbk/wBCYt/t8/8ALco+rD7kDzqLb4nozXUzduopJYWX6zmODfiuHAcJkrKiOmiF3zPDRyA4uPYBc+Cl3x1vfgEoahtFdt7sSlb0Y7sp7je/13juBt+8eSnFdHBMLjo4I6aIWjhYGt5nm49pNz4rvLl8q/1rHI2LwFodPMQFNh1VKTa1O9rf1njYb83Bb9RR9oDGRHSxUjT0p5Nt4/42c/3iP4Uxa++2KDICREXUmsIiIAiIgMre6FY6aCshqR1WPtKB60RyePgStCsrEoqSafuC4GKYdTYhTmKVolgmYC09hF2vaeBzyKrXp9oVPhU2y676d5PkJrZOHsu5OHJSfqN0uE0P4Pmd6WAXgJ9eHi3vb9D2KSMbwiCthdT1DA+N4zHEHg5p4Ec1RV2zxLXCXH75J8lPmOINwbEHI7rL19HpMZ4/JTm8rRZrz+kHb2rh0+0KmwqbZdd9O8nyE1snD2XcnDkvKAq5+C2KkiPB2K/rldcLL3k5nMrAW1cGC4eAfmtP+zRf2BRH9o3rUXdP9YlLmAfmtP8As0X9gUR/aO61F3T/AFiXP4X+T/ZN8EMBXJwwAQxACwETLdnRCpsFcXBJ2yU8MjCHNdCwtI3EbIXq6r8sTETvIiKk2SPiRgcCHAOaRYg5ghanDtF6GnlNRBSxxTOBBe0Wy42G4eC3CKSnJLSYMoiKIPlxAzOXNVZ1maQ/hCvllabxRnycHLYbx8TtHxUw659LhR0ppYnfjNU0g23xw7nO8dw8eSrmrvpmPpOx+/BGRhERWxEIiIAiIgCIiA7uFYjLTSsnhdsSxODmOHP7x2K0WgulsOKU4mZZsrbCeLjG/wD+TwKqitxovpFUYdO2op3WcMntPUkZxa4cl5MvFV8fyjKei1eOYRBWwup6hgfG8ZjiDwc08COarPp9oVPhU2y676d5PkJrZOHsu5OHJWF0M0vpsUi8pC7ZkaB5aEnpxn728itnjeEQVsLqeoYHxvGYO8Hg5p4Ec1U4+RPGn2z4+hJrZTxAvV6faFT4VNsuu+neT5Ga2Th7LuThyXk1fxnGcVKL8EC4GjEofR0z27nUsRH8sKPdfeATVEENTE0vbSmTyoGZDHhvT7hs5966WpTTlhjbhlS7Ze0nzV53PaTfyffcm3fZTC5oIIIuCMxzXPS7sW/uaJ8opapJ1W6xnYe4UtSS6iecjvdTuPrD3eY8R27HWvq1832q6hZenJvNCP0HvN9z6d26Jldp15Nf4I8FzoJmyND2OD2PaC1wN2uBzBBXKq7ardYzqBwpapxdRvd0Xb3UxJ3j3eY8R22Fhma9oexwc1wBa4G7XA5ggrn8rGlTLT4Jp7OREReUGFpdLdI4MNp3VEx3ZRsHWlfwaP8AvgmlOk1NhsJmqH2/8cY/KSu9lo+/gq0aZ6WVGKTmaY2Y24iiB6ETOQ5nmeKsMPDdr7n8phvR0dIcamrp31M7rySG/Y1vBo5ALWIsLokklpEAiIgCIiAIiIAiIgCIiA72EYpPSStnp5DFKw5Ob9DzHYVOug+t2nqQ2Gu2aafcJN0Eh7/UPfl2qvqLRfjQuWpIynouDjGFU9dA6Cdolhlb32yyc08+1Vp0+0KmwqbZdd9O8nyE1snD2XcnDkuHRnTnEMPsKecmMb4X+kiPgd3hZSGzWxQV8BpcVpHNbILOfH6RoPtgHpNI7Lrx003Y0vHxRM7TIZY8ggg2INwRvCnvVRrKFUG0Va8CoAAhlOQnGXRd7/171CuP0UMMpFNUNqYHZxSAFrtnk9rgC1wWuY8g3BsQciN4K9l9EL4aZhPRc9zQQQRcEWIO4hQPrX1amnLq2hZenNzNCP0B4ub7n07l6TVTrKFUG0Va+1SMoZTkJxu2Xe/9e9SlKW2O0Rs2zvut2qkg7cS3X6yXJTBSXqr1jOoXCkqnF1G49F291O48R7nMePfw62NGKKmk84oamEskd6SmbI1z4nHO7Wg9Ts4d26OldtQyK/K8MjwXOima5oe1wcxzbtcDdpbvuCo9021r0lGHRUpFXU2t0TeGM7uk4b+4fEKCTpHW+bik85kFM29og6zc87do7N29aleKrpkYy3N7M9xs8fxyorpTPUyGSQ7r9Vo9lo3Adi1iLCtEklpEQiIgCIiAIiIAiIgCIiAIiIAiIgCIiAIiID7Y4g3BsRuI3hck1VI/rvc7vJcuBE0gEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREB//2Q==", DB.PgDiskANN: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", @@ -88,6 +89,7 @@ def getPatternShape(i): DB.QdrantCloud.value: "#D91AD9", DB.WeaviateCloud.value: "#20C997", DB.PgVector.value: "#4C779A", + DB.PolarDBPG.value: "#2F6FA5", DB.Redis.value: "#0D6EFD", DB.AWSOpenSearch.value: "#0DCAF0", DB.OSSOpenSearch.value: "#0DCAF0", diff --git a/vectordb_bench/models.py b/vectordb_bench/models.py index 21eea9e95..81f0206be 100644 --- a/vectordb_bench/models.py +++ b/vectordb_bench/models.py @@ -177,6 +177,14 @@ class CaseConfigParamType(Enum): post_load_index = "post_load_index" pq_nbits = "pq_nbits" + # PolarDB for PostgreSQL parameters + hnsw_quantization = "hnsw_quantization" + train_samples = "train_samples" + quantization_nbits = "quantization_nbits" + graph_cache = "graph_cache" + graph_cache_timeout = "graph_cache_timeout" + iterative_scan = "iterative_scan" + # Lindorm parameters efSearch = "efSearch" pq_m = "pq_m"