From 70fc3c5724eca9f741fc20791028d350fca6617c Mon Sep 17 00:00:00 2001 From: "xichong.xc" <27044688+cxyybless@users.noreply.github.com> Date: Mon, 24 Aug 2026 11:41:30 +0800 Subject: [PATCH] feat: add support for PolarDB for PostgreSQL HNSW Add a pgvector-compatible PolarDB for PostgreSQL HNSW client to the command-line and Streamlit interfaces. Keep it separate from the existing MySQL-compatible PolarDB client. Support PQ, SQ4, SQ8, and RaBitQ index options and optional Graph Cache readiness handling. Reuse the pgvector dependency extra, add configuration and lifecycle tests, and redact connection passwords from PgVector logs. --- README.md | 43 ++ pyproject.toml | 5 +- tests/test_polardb_pg.py | 473 ++++++++++++++++++ vectordb_bench/backend/clients/__init__.py | 16 + .../backend/clients/pgvector/pgvector.py | 14 +- .../backend/clients/polardb_pg/__init__.py | 1 + .../backend/clients/polardb_pg/cli.py | 139 +++++ .../backend/clients/polardb_pg/config.py | 71 +++ .../backend/clients/polardb_pg/polardb_pg.py | 176 +++++++ vectordb_bench/cli/vectordbbench.py | 2 + .../frontend/config/dbCaseConfigs.py | 134 +++++ vectordb_bench/frontend/config/styles.py | 2 + vectordb_bench/models.py | 8 + 13 files changed, 1080 insertions(+), 4 deletions(-) create mode 100644 tests/test_polardb_pg.py create mode 100644 vectordb_bench/backend/clients/polardb_pg/__init__.py create mode 100644 vectordb_bench/backend/clients/polardb_pg/cli.py create mode 100644 vectordb_bench/backend/clients/polardb_pg/config.py create mode 100644 vectordb_bench/backend/clients/polardb_pg/polardb_pg.py diff --git a/README.md b/README.md index 0ffa98f0e..bc44b36d9 100644 --- a/README.md +++ b/README.md @@ -87,6 +87,7 @@ All the database client supported | tencent_es | `pip install vectordb-bench[tencent_es]` | | alisql | `pip install vectordb-bench[alisql]` | | polardb | `pip install vectordb-bench[polardb]` | +| polardb_pg | `pip install vectordb-bench[pgvector]` | | doris | `pip install vectordb-bench[doris]` | | zvec | `pip install vectordb-bench[zvec]` | | endee | `pip install vectordb-bench[endee]` | @@ -116,6 +117,7 @@ Options: --help Show this message and exit. Commands: + polardbpghnsw pgvectorhnsw pgvectorivfflat vectorchordrq @@ -263,6 +265,47 @@ Options: --help Show this message and exit. ``` +### Run polardb_pg from command line + +polardb_pg supports HNSW with PQ, SQ4, SQ8, RaBitQ, and Graph Cache. +This PostgreSQL client is separate from the existing MySQL-compatible PolarDB +commands and reuses the `pgvector` dependency extra: + +```shell +pip install 'vectordb-bench[pgvector]' +``` + +PolarDB-specific internal quantization and Graph Cache options require PolarDB +vector extension 0.8.3.1 or later. Use `--skip-graph-cache` when Graph Cache is +not configured on the server. + +**Example: Run hnsw index test** + +```shell +vectordbbench polardbpghnsw \ + --case-type Performance1024D1M \ + --db-label polardb-pg-rabitq8 \ + --user-name postgres --password '' \ + --host localhost --port 5432 --db-name vectordb \ + --m 16 --ef-construction 256 --ef-search 200 \ + --quantization rabitq --quantization-nbits 8 \ + --iterative-scan off --graph-cache +``` + +To list the options for polardb_pg, execute +`vectordbbench polardbpghnsw --help`. The following are some PolarDB-specific +command-line options. + +```text + --quantization [pq|sq4|sq8|rabitq] + --pq-m INTEGER + --train-samples INTEGER + --quantization-nbits [1|4|8] + --graph-cache / --skip-graph-cache + --graph-cache-timeout INTEGER + --iterative-scan [off|strict_order|relaxed_order] +``` + ### Run VectorChord (vchordrq) from command line VectorChord is a PostgreSQL extension for scalable vector similarity search using IVF + RaBitQ indexing. diff --git a/pyproject.toml b/pyproject.toml index 9fbbb7748..43f793c60 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -54,6 +54,9 @@ test = [ "black", "ruff", "pytest", + "psycopg", + "psycopg-binary", + "pgvector", ] restful = [ "flask" ] qdrant = [ "qdrant-client" ] @@ -63,7 +66,7 @@ elastic = [ "elasticsearch" ] # For elastic and aliyun_elasticsearch pgvector = [ "psycopg", "psycopg-binary", "pgvector" ] -# for pgvector, pgvectorscale, pgdiskann, alloydb, and lakebase_vector +# for pgvector, pgvectorscale, pgdiskann, alloydb, lakebase_vector, and PolarDB for PostgreSQL HNSW pgvecto_rs = [ "pgvecto_rs[psycopg3]>=0.2.2" ] redis = [ "redis" ] diff --git a/tests/test_polardb_pg.py b/tests/test_polardb_pg.py new file mode 100644 index 000000000..4d8fdc58b --- /dev/null +++ b/tests/test_polardb_pg.py @@ -0,0 +1,473 @@ +from __future__ import annotations + +from contextlib import contextmanager +from typing import TYPE_CHECKING +from unittest.mock import MagicMock, call + +import pytest +from click.testing import CliRunner +from pydantic import ValidationError + +from vectordb_bench.backend.cases import CaseLabel +from vectordb_bench.backend.clients import DB +from vectordb_bench.backend.clients.api import IndexType +from vectordb_bench.backend.clients.pgvector.pgvector import PgVector +from vectordb_bench.backend.clients.polardb_pg.cli import PolarDBPgHNSW as PolarDBPgHNSWCommand +from vectordb_bench.backend.clients.polardb_pg.config import PolarDBPgHNSWConfig +from vectordb_bench.backend.clients.polardb_pg.polardb_pg import ( + HNSWGraphCacheDetail, + PolarDBPgHNSW, +) +from vectordb_bench.frontend.config.dbCaseConfigs import get_case_config_inputs +from vectordb_bench.models import CaseConfigParamType + +if TYPE_CHECKING: + from collections.abc import Iterator + + +def make_config(**kwargs) -> PolarDBPgHNSWConfig: + return PolarDBPgHNSWConfig( + metric_type="COSINE", + m=16, + ef_construction=256, + ef_search=200, + **kwargs, + ) + + +def cache_detail( + status: str, + *, + requested: str = "on", + usable: bool = False, + last_error: str | None = None, +) -> HNSWGraphCacheDetail: + detail = f"requested={requested} status={status} usable={'yes' if usable else 'no'}" + if last_error is not None: + detail += f" last_error={last_error}" + return HNSWGraphCacheDetail.parse(detail) + + +def make_client(details: list[HNSWGraphCacheDetail], **config_kwargs) -> PolarDBPgHNSW: + client = object.__new__(PolarDBPgHNSW) + client.case_config = make_config(**config_kwargs) + client._index_name = "pgvector_index" + client._check_graph_cache_role = MagicMock() + client._graph_cache_detail = MagicMock(side_effect=details) + client._call_graph_cache_function = MagicMock(return_value=True) + return client + + +class TestPolarDBPgConfig: + def test_rabitq_graph_cache_index_options(self): + config = make_config( + quantization="rabitq", + quantization_nbits=8, + train_samples=100000, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert options == { + "m": "16", + "ef_construction": "256", + "quantization": "rabitq", + "train_samples": "100000", + "quantization_nbits": "8", + "cache": "on", + } + + def test_frontend_quantization_key_maps_to_index_option(self): + config = make_config( + hnsw_quantization="rabitq", + quantization_nbits=8, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert config.quantization == "rabitq" + assert options["quantization"] == "rabitq" + assert options["quantization_nbits"] == "8" + + def test_pq_index_options(self): + config = make_config( + graph_cache=False, + quantization="pq", + pq_m=32, + train_samples=5000, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert options == { + "m": "16", + "ef_construction": "256", + "quantization": "pq", + "pq_m": "32", + "train_samples": "5000", + } + + @pytest.mark.parametrize("quantization", ["sq4", "sq8"]) + def test_sq_index_options(self, quantization: str): + config = make_config( + graph_cache=False, + quantization=quantization, + train_samples=5000, + ) + + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + + assert options == { + "m": "16", + "ef_construction": "256", + "quantization": quantization, + "train_samples": "5000", + } + + def test_graph_cache_can_be_disabled(self): + config = make_config(graph_cache=False, iterative_scan="relaxed_order") + options = {item["option_name"]: item["val"] for item in config.index_param()["index_creation_with_options"]} + assert "cache" not in options + + def test_graph_cache_timeout_must_be_positive(self): + assert make_config().graph_cache_timeout == 3600 + with pytest.raises(ValidationError): + make_config(graph_cache_timeout=0) + + @pytest.mark.parametrize("nbits", [1, 4, 8]) + def test_supported_rabitq_nbits(self, nbits: int): + assert make_config(quantization="rabitq", quantization_nbits=nbits).quantization_nbits == nbits + + @pytest.mark.parametrize("nbits", [0, 2, 9]) + def test_rejects_unsupported_rabitq_nbits(self, nbits: int): + with pytest.raises(ValidationError): + make_config(quantization="rabitq", quantization_nbits=nbits) + + def test_rejects_iterative_scan_with_graph_cache(self): + with pytest.raises(ValidationError, match="Graph Cache requires"): + make_config(iterative_scan="strict_order") + + def test_frontend_post_load_switch_controls_index_timing(self): + before_load = make_config(post_load_index=False) + after_load = make_config(post_load_index=True) + + assert before_load.create_index_before_load is True + assert before_load.create_index_after_load is False + assert after_load.create_index_before_load is False + assert after_load.create_index_after_load is True + + def test_rejects_quantizer_specific_options(self): + with pytest.raises(ValidationError, match="pq_m"): + make_config(quantization="sq8", pq_m=16) + with pytest.raises(ValidationError, match="quantization_nbits"): + make_config(quantization="pq", quantization_nbits=8) + + def test_rejects_opq(self): + with pytest.raises(ValidationError): + make_config(quantization="opq") + + def test_connection_password_is_redacted(self, monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture): + connection = MagicMock() + cursor = MagicMock() + monkeypatch.setattr(PgVector, "_create_connection", MagicMock(return_value=(connection, cursor))) + caplog.set_level("INFO", logger="vectordb_bench.backend.clients.pgvector.pgvector") + + PgVector( + dim=8, + db_config={ + "connect_config": { + "host": "localhost", + "port": 5432, + "dbname": "vectordb", + "user": "postgres", + "password": "do-not-log-this-password", + }, + "table_name": "vdbbench_table_test", + }, + db_case_config=make_config(graph_cache=False), + ) + + assert "do-not-log-this-password" not in caplog.text + assert "**********" in caplog.text + + def test_polardb_name_is_used_during_base_initialization( + self, + monkeypatch: pytest.MonkeyPatch, + caplog: pytest.LogCaptureFixture, + ): + connection = MagicMock() + cursor = MagicMock() + monkeypatch.setattr(PgVector, "_create_connection", MagicMock(return_value=(connection, cursor))) + caplog.set_level("INFO", logger="vectordb_bench.backend.clients.pgvector.pgvector") + + client = PolarDBPgHNSW( + dim=8, + db_config={ + "connect_config": { + "host": "localhost", + "port": 5432, + "dbname": "vectordb", + "user": "postgres", + "password": "secret", + }, + "table_name": "vdbbench_table_test", + }, + db_case_config=make_config(graph_cache=False), + ) + + assert client.name == "PolarDBPG" + assert "PolarDBPG config values" in caplog.text + assert "PgVector config values" not in caplog.text + + +class TestGraphCacheDetail: + def test_parse_ready_detail(self): + detail = HNSWGraphCacheDetail.parse( + "requested=on dbid=1 index_oid=2 status=ready usable=yes graph_bytes=100 last_error=none", + ) + assert detail.requested == "on" + assert detail.status == "ready" + assert detail.usable is True + assert detail.last_error is None + + def test_parse_error_with_spaces(self): + detail = HNSWGraphCacheDetail.parse( + "requested=on status=empty usable=no last_error=could not read graph page", + ) + assert detail.last_error == "could not read graph page" + + def test_rejects_incomplete_detail(self): + with pytest.raises(RuntimeError, match="missing usable"): + HNSWGraphCacheDetail.parse("requested=on status=empty") + + +class TestGraphCacheLifecycle: + def test_search_only_initialization_waits_for_cache(self, monkeypatch: pytest.MonkeyPatch): + def base_init( + client: PolarDBPgHNSW, + *_args: object, + db_case_config: PolarDBPgHNSWConfig, + **_kwargs: object, + ) -> None: + client.case_config = db_case_config + client._index_name = "pgvector_index" + + @contextmanager + def fake_init(_client: PolarDBPgHNSW) -> Iterator[None]: + yield + + wait_for_cache = MagicMock() + monkeypatch.setattr(PgVector, "__init__", base_init) + monkeypatch.setattr(PolarDBPgHNSW, "init", fake_init) + monkeypatch.setattr(PolarDBPgHNSW, "_ensure_graph_cache_ready", wait_for_cache) + + PolarDBPgHNSW(db_case_config=make_config(), drop_old=False) + + wait_for_cache.assert_called_once_with() + + def test_post_insert_waits_after_index_creation(self, monkeypatch: pytest.MonkeyPatch): + create_index = MagicMock() + wait_for_cache = MagicMock() + client = object.__new__(PolarDBPgHNSW) + client.case_config = make_config() + monkeypatch.setattr(PgVector, "_post_insert", create_index) + monkeypatch.setattr(PolarDBPgHNSW, "_ensure_graph_cache_ready", wait_for_cache) + + client._post_insert() + + create_index.assert_called_once_with() + wait_for_cache.assert_called_once_with() + + def test_ready_cache_needs_no_lifecycle_call(self): + client = make_client([cache_detail("ready", usable=True)]) + + client._ensure_graph_cache_ready() + + client._check_graph_cache_role.assert_called_once_with() + client._call_graph_cache_function.assert_not_called() + + def test_empty_cache_is_scheduled_before_ready(self, monkeypatch: pytest.MonkeyPatch): + client = make_client( + [ + cache_detail("empty"), + cache_detail("building"), + cache_detail("ready", usable=True), + ], + ) + monkeypatch.setattr("vectordb_bench.backend.clients.polardb_pg.polardb_pg.time.sleep", lambda _: None) + + client._ensure_graph_cache_ready() + + client._call_graph_cache_function.assert_called_once_with("hnsw_schedule_cache_rebuild") + + def test_stale_cache_is_released_and_rebuilt(self, monkeypatch: pytest.MonkeyPatch): + client = make_client( + [ + cache_detail("ready", usable=False), + cache_detail("draining"), + cache_detail("empty"), + cache_detail("building"), + cache_detail("ready", usable=True), + ], + ) + monkeypatch.setattr("vectordb_bench.backend.clients.polardb_pg.polardb_pg.time.sleep", lambda _: None) + + client._ensure_graph_cache_ready() + + assert client._call_graph_cache_function.call_args_list == [ + call("hnsw_release_cache"), + call("hnsw_schedule_cache_rebuild"), + ] + + def test_not_preloaded_fails_immediately(self): + client = make_client([cache_detail("not_preloaded")]) + with pytest.raises(RuntimeError, match="shared_preload_libraries"): + client._ensure_graph_cache_ready() + + def test_index_without_cache_reloption_fails(self): + client = make_client([cache_detail("empty", requested="off")]) + with pytest.raises(RuntimeError, match="cache=on"): + client._ensure_graph_cache_ready() + + def test_build_error_fails_without_retry(self): + client = make_client([cache_detail("empty", last_error="out of shared memory")]) + with pytest.raises(RuntimeError, match="out of shared memory"): + client._ensure_graph_cache_ready() + client._call_graph_cache_function.assert_not_called() + + def test_timeout_reports_last_state(self, monkeypatch: pytest.MonkeyPatch): + client = make_client([cache_detail("building")], graph_cache_timeout=1) + monotonic = MagicMock(side_effect=[0.0, 2.0]) + monkeypatch.setattr("vectordb_bench.backend.clients.polardb_pg.polardb_pg.time.monotonic", monotonic) + + with pytest.raises(TimeoutError, match="after 1s.*status=building"): + client._ensure_graph_cache_ready() + + +def test_polardb_pg_cli_help(): + result = CliRunner().invoke(PolarDBPgHNSWCommand, ["--help"]) + + assert result.exit_code == 0, result.output + assert PolarDBPgHNSWCommand.name == "polardbpghnsw" + assert "--graph-cache / --skip-graph-cache" in result.output + assert "--graph-cache-poll-interval" not in result.output + assert "--graph-cache-timeout INTEGER" in result.output + assert "--quantization [pq|sq4|sq8|rabitq]" in result.output + assert "--quantization-nbits [1|4|8]" in result.output + + +def test_polardb_pg_cli_dry_run_builds_task( + monkeypatch: pytest.MonkeyPatch, + caplog: pytest.LogCaptureFixture, +): + benchmark_run = MagicMock() + monkeypatch.setattr("vectordb_bench.cli.cli.benchmark_runner.run", benchmark_run) + caplog.set_level("INFO", logger="vectordb_bench.cli.cli") + + result = CliRunner().invoke( + PolarDBPgHNSWCommand, + [ + "--case-type", + "Performance1024D1M", + "--db-label", + "polardb-pg-dry-run", + "--user-name", + "postgres", + "--password", + "secret", + "--host", + "localhost", + "--port", + "5432", + "--db-name", + "vec", + "--m", + "16", + "--ef-construction", + "256", + "--ef-search", + "200", + "--maintenance-work-mem", + "128GB", + "--max-parallel-workers", + "64", + "--quantization-type", + "none", + "--table-quantization-type", + "none", + "--skip-reranking", + "--quantized-fetch-limit", + "400", + "--iterative-scan", + "off", + "--skip-create-index-before-load", + "--create-index-after-load", + "--graph-cache", + "--graph-cache-timeout", + "120", + "--quantization", + "rabitq", + "--train-samples", + "5000", + "--quantization-nbits", + "8", + "--dry-run", + ], + ) + + assert result.exit_code == 0, result.output + benchmark_run.assert_not_called() + assert "graph_cache_timeout=120" in caplog.text + assert "quantization='rabitq'" in caplog.text + assert "secret" not in caplog.text + + +class TestPolarDBPgFrontend: + @staticmethod + def input_map(case_label: CaseLabel): + return {config.label: config for config in get_case_config_inputs(DB.PolarDBPG, case_label)} + + def test_uses_hnsw_only_and_exposes_polardb_options(self): + load_inputs = self.input_map(CaseLabel.Load) + performance_inputs = self.input_map(CaseLabel.Performance) + + assert load_inputs[CaseConfigParamType.IndexType].inputConfig["options"] == [IndexType.HNSW.value] + assert CaseConfigParamType.hnsw_quantization in load_inputs + assert CaseConfigParamType.quantization_nbits in load_inputs + assert CaseConfigParamType.graph_cache in load_inputs + assert CaseConfigParamType.graph_cache_timeout in load_inputs + assert CaseConfigParamType.post_load_index in load_inputs + assert CaseConfigParamType.ef_search in performance_inputs + assert CaseConfigParamType.graph_cache_timeout in performance_inputs + assert CaseConfigParamType.iterative_scan in performance_inputs + + def test_quantization_parameter_has_a_distinct_enum_value(self): + assert CaseConfigParamType.hnsw_quantization.value == "hnsw_quantization" + assert CaseConfigParamType.hnsw_quantization is not CaseConfigParamType.mongodb_quantization_type + + def test_quantizer_fields_are_conditionally_displayed(self): + inputs = self.input_map(CaseLabel.Performance) + config = { + CaseConfigParamType.IndexType: IndexType.HNSW.value, + CaseConfigParamType.hnsw_quantization: "rabitq", + CaseConfigParamType.graph_cache: True, + } + + assert inputs[CaseConfigParamType.train_samples].isDisplayed(config) + assert inputs[CaseConfigParamType.quantization_nbits].isDisplayed(config) + assert not inputs[CaseConfigParamType.pq_m].isDisplayed(config) + + config[CaseConfigParamType.hnsw_quantization] = "pq" + assert inputs[CaseConfigParamType.pq_m].isDisplayed(config) + assert not inputs[CaseConfigParamType.quantization_nbits].isDisplayed(config) + + def test_iterative_scan_is_hidden_with_graph_cache(self): + inputs = self.input_map(CaseLabel.Performance) + config = {CaseConfigParamType.graph_cache: True} + + assert inputs[CaseConfigParamType.graph_cache_timeout].isDisplayed(config) + assert not inputs[CaseConfigParamType.iterative_scan].isDisplayed(config) + + config[CaseConfigParamType.graph_cache] = False + assert not inputs[CaseConfigParamType.graph_cache_timeout].isDisplayed(config) + assert inputs[CaseConfigParamType.iterative_scan].isDisplayed(config) diff --git a/vectordb_bench/backend/clients/__init__.py b/vectordb_bench/backend/clients/__init__.py index dbd83bb66..cfc4bf15c 100644 --- a/vectordb_bench/backend/clients/__init__.py +++ b/vectordb_bench/backend/clients/__init__.py @@ -30,6 +30,7 @@ class DB(Enum): QdrantLocal = "QdrantLocal" WeaviateCloud = "WeaviateCloud" PgVector = "PgVector" + PolarDBPG = "PolarDBPG" PgVectoRS = "PgVectoRS" PgVectorScale = "PgVectorScale" PgDiskANN = "PgDiskANN" @@ -110,6 +111,11 @@ def init_cls(self) -> type[VectorDB]: # noqa: PLR0911, PLR0912, C901, PLR0915 return PgVector + if self == DB.PolarDBPG: + from .polardb_pg.polardb_pg import PolarDBPgHNSW + + return PolarDBPgHNSW + if self == DB.PgVectoRS: from .pgvecto_rs.pgvecto_rs import PgVectoRS @@ -333,6 +339,11 @@ def config_cls(self) -> type[DBConfig]: # noqa: PLR0911, PLR0912, C901, PLR0915 return PgVectorConfig + if self == DB.PolarDBPG: + from .polardb_pg.config import PolarDBPgConfig + + return PolarDBPgConfig + if self == DB.PgVectoRS: from .pgvecto_rs.config import PgVectoRSConfig @@ -560,6 +571,11 @@ def case_config_cls( # noqa: C901, PLR0911, PLR0912, PLR0915 return _pgvector_case_config.get(index_type) + if self == DB.PolarDBPG: + from .polardb_pg.config import _polardb_pg_case_config + + return _polardb_pg_case_config.get(index_type) + if self == DB.PgVectoRS: from .pgvecto_rs.config import _pgvecto_rs_case_config diff --git a/vectordb_bench/backend/clients/pgvector/pgvector.py b/vectordb_bench/backend/clients/pgvector/pgvector.py index 13f471afe..c60b5d69b 100644 --- a/vectordb_bench/backend/clients/pgvector/pgvector.py +++ b/vectordb_bench/backend/clients/pgvector/pgvector.py @@ -19,9 +19,17 @@ log = logging.getLogger(__name__) +def _redact_connect_config(connect_config: dict[str, Any]) -> dict[str, Any]: + redacted = dict(connect_config) + if "password" in redacted: + redacted["password"] = "**********" # noqa: S105 + return redacted + + class PgVector(VectorDB): """Use psycopg instructions""" + name = "PgVector" thread_safe: bool = False supported_filter_types: list[FilterOp] = [ FilterOp.NonFilter, @@ -43,7 +51,6 @@ def __init__( with_scalar_labels: bool = False, **kwargs, ): - self.name = "PgVector" self.case_config = db_case_config self.table_name = db_config["table_name"] self.connect_config = db_config["connect_config"] @@ -62,7 +69,8 @@ def __init__( self.cursor.execute("CREATE EXTENSION IF NOT EXISTS vector") self.conn.commit() - log.info(f"{self.name} config values: {self.connect_config}\n{self.case_config}") + redacted_connect_config = _redact_connect_config(self.connect_config) + log.info("%s config values: %s\n%s", self.name, redacted_connect_config, self.case_config) if not any( ( self.case_config.create_index_before_load, @@ -71,7 +79,7 @@ def __init__( ): msg = ( f"{self.name} config must create an index using create_index_before_load or create_index_after_load" - f"{self.name} config values: {self.connect_config}\n{self.case_config}" + f"{self.name} config values: {redacted_connect_config}\n{self.case_config}" ) log.error(msg) raise RuntimeError(msg) diff --git a/vectordb_bench/backend/clients/polardb_pg/__init__.py b/vectordb_bench/backend/clients/polardb_pg/__init__.py new file mode 100644 index 000000000..10a4ff541 --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/__init__.py @@ -0,0 +1 @@ +"""PolarDB for PostgreSQL vector benchmark client.""" diff --git a/vectordb_bench/backend/clients/polardb_pg/cli.py b/vectordb_bench/backend/clients/polardb_pg/cli.py new file mode 100644 index 000000000..8fd737074 --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/cli.py @@ -0,0 +1,139 @@ +from typing import Annotated, TypedDict, Unpack + +import click +from pydantic import SecretStr + +from vectordb_bench.backend.clients import DB + +from ....cli.cli import ( + HNSWFlavor1, + cli, + click_parameter_decorators_from_typed_dict, + get_custom_case_config, + run, +) +from ..pgvector.cli import PgVectorTypedDict +from .config import DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS + + +def parse_quantization_nbits(ctx: object, param: object, value: str | None) -> int | None: # noqa: ARG001 + return int(value) if value is not None else None + + +class PolarDBPgHNSWOptions(TypedDict): + iterative_scan: Annotated[ + str, + click.option( + "--iterative-scan", + type=click.Choice(["off", "strict_order", "relaxed_order"]), + default="off", + show_default=True, + help="HNSW iterative scan mode; Graph Cache requires off", + ), + ] + create_index_before_load: Annotated[ + bool, + click.option( + "--create-index-before-load/--skip-create-index-before-load", + default=False, + show_default=True, + help="Create the HNSW index before loading data", + ), + ] + create_index_after_load: Annotated[ + bool, + click.option( + "--create-index-after-load/--skip-create-index-after-load", + default=True, + show_default=True, + help="Create the HNSW index after loading data", + ), + ] + graph_cache: Annotated[ + bool, + click.option( + "--graph-cache/--skip-graph-cache", + default=True, + show_default=True, + help="Build Graph Cache and wait until it is usable before search", + ), + ] + graph_cache_timeout: Annotated[ + int, + click.option( + "--graph-cache-timeout", + type=click.IntRange(min=1), + default=DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS, + show_default=True, + help="Seconds to wait for Graph Cache to become usable", + ), + ] + quantization: Annotated[ + str | None, + click.option( + "--quantization", + type=click.Choice(["pq", "sq4", "sq8", "rabitq"]), + help="PolarDB HNSW internal quantization method", + ), + ] + pq_m: Annotated[ + int | None, + click.option("--pq-m", type=int, help="Number of PQ sub-quantizers"), + ] + train_samples: Annotated[ + int | None, + click.option("--train-samples", type=int, help="Number of quantizer training samples"), + ] + quantization_nbits: Annotated[ + int | None, + click.option( + "--quantization-nbits", + type=click.Choice(["1", "4", "8"]), + callback=parse_quantization_nbits, + help="RaBitQ bits per dimension", + ), + ] + + +class PolarDBPgHNSWTypedDict(PgVectorTypedDict, HNSWFlavor1, PolarDBPgHNSWOptions): ... + + +@cli.command() +@click_parameter_decorators_from_typed_dict(PolarDBPgHNSWTypedDict) +def PolarDBPgHNSW(**parameters: Unpack[PolarDBPgHNSWTypedDict]): + from .config import PolarDBPgConfig, PolarDBPgHNSWConfig + + parameters["custom_case"] = get_custom_case_config(parameters) + run( + db=DB.PolarDBPG, + db_config=PolarDBPgConfig( + db_label=parameters["db_label"], + user_name=SecretStr(parameters["user_name"]), + password=SecretStr(parameters["password"]), + host=parameters["host"], + port=parameters["port"], + db_name=parameters["db_name"], + ), + db_case_config=PolarDBPgHNSWConfig( + m=parameters["m"], + ef_construction=parameters["ef_construction"], + ef_search=parameters["ef_search"], + maintenance_work_mem=parameters["maintenance_work_mem"], + max_parallel_workers=parameters["max_parallel_workers"], + quantization_type=parameters["quantization_type"], + table_quantization_type=parameters["table_quantization_type"], + reranking=parameters["reranking"], + reranking_metric=parameters["reranking_metric"], + quantized_fetch_limit=parameters["quantized_fetch_limit"], + iterative_scan=parameters["iterative_scan"], + create_index_before_load=parameters["create_index_before_load"], + create_index_after_load=parameters["create_index_after_load"], + graph_cache=parameters["graph_cache"], + graph_cache_timeout=parameters["graph_cache_timeout"], + quantization=parameters["quantization"], + pq_m=parameters["pq_m"], + train_samples=parameters["train_samples"], + quantization_nbits=parameters["quantization_nbits"], + ), + **parameters, + ) diff --git a/vectordb_bench/backend/clients/polardb_pg/config.py b/vectordb_bench/backend/clients/polardb_pg/config.py new file mode 100644 index 000000000..f1e12bc35 --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/config.py @@ -0,0 +1,71 @@ +from typing import Literal + +from pydantic import AliasChoices, Field, PositiveInt, model_validator + +from ..api import IndexType +from ..pgvector.config import PgVectorConfig, PgVectorHNSWConfig, PgVectorIndexParam + +DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS = 3600 + + +class PolarDBPgConfig(PgVectorConfig): + """Connection configuration for PolarDB for PostgreSQL.""" + + +class PolarDBPgHNSWConfig(PgVectorHNSWConfig): + """PolarDB HNSW-specific benchmark options.""" + + post_load_index: bool | None = None + iterative_scan: str = "off" + graph_cache: bool = True + graph_cache_timeout: PositiveInt = DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS + quantization: Literal["pq", "sq4", "sq8", "rabitq"] | None = Field( + default=None, + validation_alias=AliasChoices("quantization", "hnsw_quantization"), + ) + pq_m: int | None = None + train_samples: int | None = None + quantization_nbits: Literal[1, 4, 8] | None = None + + @model_validator(mode="after") + def validate_polardb_options(self) -> "PolarDBPgHNSWConfig": + if self.post_load_index is not None: + self.create_index_before_load = not self.post_load_index + self.create_index_after_load = self.post_load_index + if self.iterative_scan not in {"off", "strict_order", "relaxed_order"}: + msg = "iterative_scan must be one of: off, strict_order, relaxed_order" + raise ValueError(msg) + if self.graph_cache and self.iterative_scan != "off": + msg = "Graph Cache requires iterative_scan=off" + raise ValueError(msg) + if self.pq_m is not None and self.quantization != "pq": + msg = "pq_m is only valid with quantization=pq" + raise ValueError(msg) + if self.quantization_nbits is not None and self.quantization != "rabitq": + msg = "quantization_nbits is only valid with quantization=rabitq" + raise ValueError(msg) + if self.train_samples is not None and self.quantization is None: + msg = "train_samples requires a quantization method" + raise ValueError(msg) + return self + + def index_param(self) -> PgVectorIndexParam: + index_param = super().index_param() + polar_options = { + "quantization": self.quantization, + "pq_m": self.pq_m, + "train_samples": self.train_samples, + "quantization_nbits": self.quantization_nbits, + "cache": "on" if self.graph_cache else None, + } + index_param["index_creation_with_options"] = [ + *index_param["index_creation_with_options"], + *self._optionally_build_with_options(polar_options), + ] + return index_param + + +_polardb_pg_case_config = { + IndexType.HNSW: PolarDBPgHNSWConfig, + IndexType.ES_HNSW: PolarDBPgHNSWConfig, +} diff --git a/vectordb_bench/backend/clients/polardb_pg/polardb_pg.py b/vectordb_bench/backend/clients/polardb_pg/polardb_pg.py new file mode 100644 index 000000000..fe50261bd --- /dev/null +++ b/vectordb_bench/backend/clients/polardb_pg/polardb_pg.py @@ -0,0 +1,176 @@ +"""PolarDB for PostgreSQL HNSW client.""" + +from __future__ import annotations + +import logging +import re +import time +from dataclasses import dataclass +from typing import TYPE_CHECKING + +from psycopg import sql + +from ..pgvector.pgvector import PgVector + +if TYPE_CHECKING: + from .config import PolarDBPgHNSWConfig + +log = logging.getLogger(__name__) + +_GRAPH_CACHE_POLL_INTERVAL_SECONDS = 1.0 + + +@dataclass(frozen=True) +class HNSWGraphCacheDetail: + requested: str + status: str + usable: bool + last_error: str | None + raw: str + + @staticmethod + def _field(detail: str, name: str) -> str: + match = re.search(rf"(?:^|\s){re.escape(name)}=([^\s]+)", detail) + if match is None: + msg = f"Graph Cache detail is missing {name}: {detail}" + raise RuntimeError(msg) + return match.group(1) + + @classmethod + def parse(cls, detail: str) -> HNSWGraphCacheDetail: + requested = cls._field(detail, "requested") + status = cls._field(detail, "status") + usable_value = cls._field(detail, "usable") + if usable_value not in {"yes", "no"}: + msg = f"Graph Cache detail has invalid usable state: {detail}" + raise RuntimeError(msg) + error_match = re.search(r"(?:^|\s)last_error=(.*)$", detail) + last_error = error_match.group(1) if error_match else None + if last_error == "none": + last_error = None + return cls( + requested=requested, + status=status, + usable=usable_value == "yes", + last_error=last_error, + raw=detail, + ) + + +class PolarDBPgHNSW(PgVector): + """PgVector-compatible PolarDB HNSW client.""" + + name = "PolarDBPG" + case_config: PolarDBPgHNSWConfig + + def __init__(self, *args, drop_old: bool = False, **kwargs): + super().__init__(*args, drop_old=drop_old, **kwargs) + + # Search-only benchmarks skip optimize(), so validate and prepare the + # existing index before any search workers can start. + if not drop_old and self.case_config.graph_cache: + with self.init(): + self._ensure_graph_cache_ready() + + def _post_insert(self): + super()._post_insert() + if self.case_config.graph_cache: + self._ensure_graph_cache_ready() + + def _qualified_index_name(self) -> str: + return f"public.{self._index_name}" + + def _graph_cache_detail(self) -> HNSWGraphCacheDetail: + assert self.conn is not None, "Connection is not initialized" + assert self.cursor is not None, "Cursor is not initialized" + + row = self.cursor.execute( + "SELECT hnsw_cache_detail(%s::regclass)", + (self._qualified_index_name(),), + ).fetchone() + self.conn.commit() + if row is None or row[0] is None: + msg = f"Unable to read Graph Cache state for {self._qualified_index_name()}" + raise RuntimeError(msg) + return HNSWGraphCacheDetail.parse(row[0]) + + def _call_graph_cache_function(self, function_name: str) -> bool: + assert self.conn is not None, "Connection is not initialized" + assert self.cursor is not None, "Cursor is not initialized" + + statement = sql.SQL("SELECT {}(%s::regclass)").format(sql.Identifier(function_name)) + row = self.cursor.execute(statement, (self._qualified_index_name(),)).fetchone() + self.conn.commit() + return bool(row and row[0]) + + def _check_graph_cache_role(self): + assert self.conn is not None, "Connection is not initialized" + assert self.cursor is not None, "Cursor is not initialized" + + row = self.cursor.execute("SELECT pg_is_in_recovery()").fetchone() + self.conn.commit() + if row and row[0]: + msg = "PolarDB HNSW Graph Cache benchmarks must run on the primary node" + raise RuntimeError(msg) + + def _ensure_graph_cache_ready(self): + self._check_graph_cache_role() + timeout = self.case_config.graph_cache_timeout + deadline = time.monotonic() + timeout + last_logged_state: tuple[str, bool] | None = None + last_progress_log = 0.0 + release_requested = False + + while True: + detail = self._graph_cache_detail() + if detail.requested != "on": + msg = ( + f"HNSW index {self._qualified_index_name()} was not created with cache=on; " + "enable that reloption, rebuild the index, or run with --skip-graph-cache" + ) + raise RuntimeError(msg) + if detail.status == "not_preloaded": + msg = ( + "HNSW Graph Cache is not preloaded; add vector to " + "shared_preload_libraries and restart PostgreSQL" + ) + raise RuntimeError(msg) + if detail.status == "ready" and detail.usable: + log.info("HNSW Graph Cache is ready for %s", self._qualified_index_name()) + return + now = time.monotonic() + state = (detail.status, detail.usable) + if state != last_logged_state or now - last_progress_log >= 60: + log.info( + "Waiting for HNSW Graph Cache: index=%s status=%s usable=%s", + self._qualified_index_name(), + detail.status, + "yes" if detail.usable else "no", + ) + last_logged_state = state + last_progress_log = now + + if detail.status == "ready": + if not release_requested: + log.info("Rebuilding stale HNSW Graph Cache for %s", self._qualified_index_name()) + self._call_graph_cache_function("hnsw_release_cache") + release_requested = True + elif detail.status == "empty": + if detail.last_error: + msg = f"HNSW Graph Cache build failed for {self._qualified_index_name()}: {detail.last_error}" + raise RuntimeError(msg) + accepted = self._call_graph_cache_function("hnsw_schedule_cache_rebuild") + if accepted: + release_requested = False + log.info("Scheduled HNSW Graph Cache build for %s", self._qualified_index_name()) + elif detail.status not in {"building", "draining"}: + msg = f"Unsupported HNSW Graph Cache state for {self._qualified_index_name()}: {detail.raw}" + raise RuntimeError(msg) + + if now >= deadline: + msg = ( + f"Timed out after {timeout:g}s waiting for " + f"HNSW Graph Cache on {self._qualified_index_name()}: {detail.raw}" + ) + raise TimeoutError(msg) + time.sleep(min(_GRAPH_CACHE_POLL_INTERVAL_SECONDS, max(deadline - now, 0))) diff --git a/vectordb_bench/cli/vectordbbench.py b/vectordb_bench/cli/vectordbbench.py index 1bbc462ef..e4d274a52 100644 --- a/vectordb_bench/cli/vectordbbench.py +++ b/vectordb_bench/cli/vectordbbench.py @@ -41,6 +41,7 @@ PolarDBHNSWPQ, PolarDBHNSWSQ, ) +from ..backend.clients.polardb_pg.cli import PolarDBPgHNSW from ..backend.clients.qdrant_cloud.cli import QdrantCloud from ..backend.clients.qdrant_local.cli import QdrantLocal from ..backend.clients.redis.cli import Redis @@ -61,6 +62,7 @@ cli.add_command(AdbpgNova) cli.add_command(PgVectorHNSW) +cli.add_command(PolarDBPgHNSW) cli.add_command(PgVectoRSHNSW) cli.add_command(PgVectoRSIVFFlat) cli.add_command(Redis) diff --git a/vectordb_bench/frontend/config/dbCaseConfigs.py b/vectordb_bench/frontend/config/dbCaseConfigs.py index 678b53288..f23ff3614 100644 --- a/vectordb_bench/frontend/config/dbCaseConfigs.py +++ b/vectordb_bench/frontend/config/dbCaseConfigs.py @@ -6,6 +6,7 @@ from vectordb_bench.backend.cases import CaseLabel, CaseType, PerformanceCase from vectordb_bench.backend.clients import DB from vectordb_bench.backend.clients.api import IndexType, MetricType, SQType +from vectordb_bench.backend.clients.polardb_pg.config import DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS from vectordb_bench.backend.dataset import ( LAION_INT_FILTER_SEARCH_WIDTHS, DatasetManager, @@ -1594,6 +1595,96 @@ class CaseConfigInput(BaseModel): ) +# PolarDB for PostgreSQL HNSW configs +CaseConfigParamInput_IndexType_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.IndexType, + inputHelp="PolarDB for PostgreSQL currently supports the HNSW index in VectorDBBench", + inputType=InputType.Option, + inputConfig={"options": [IndexType.HNSW.value]}, +) + +CaseConfigParamInput_HnswQuantization_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.hnsw_quantization, + displayLabel="Internal quantization", + inputHelp="PolarDB HNSW internal quantizer; None keeps the standard pgvector code format", + inputType=InputType.Option, + inputConfig={"options": [None, "pq", "sq4", "sq8", "rabitq"]}, +) + +CaseConfigParamInput_PostLoadIndex_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.post_load_index, + displayLabel="Create index after load", + inputHelp="Disable to build the HNSW index before loading data", + inputType=InputType.Bool, + inputConfig={"value": True}, +) + +CaseConfigParamInput_PQM_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.pq_m, + displayLabel="PQ sub-quantizers", + inputHelp="Number of sub-quantizers used by PQ", + inputType=InputType.Number, + inputConfig={ + "min": 2, + "max": 4096, + "value": 32, + }, + isDisplayed=lambda config: config.get(CaseConfigParamType.hnsw_quantization, None) == "pq", +) + +CaseConfigParamInput_TrainSamples_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.train_samples, + displayLabel="Quantizer train samples", + inputHelp="Number of sampled rows used to train the internal quantizer", + inputType=InputType.Number, + inputConfig={ + "min": 100, + "max": MAX_STREAMLIT_INT, + "value": 1000, + }, + isDisplayed=lambda config: config.get(CaseConfigParamType.hnsw_quantization, None) is not None, +) + +CaseConfigParamInput_QuantizationNbits_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.quantization_nbits, + displayLabel="RaBitQ bits per dimension", + inputHelp="RaBitQ code width", + inputType=InputType.Option, + inputConfig={"options": [1, 4, 8]}, + isDisplayed=lambda config: config.get(CaseConfigParamType.hnsw_quantization, None) == "rabitq", +) + +CaseConfigParamInput_GraphCache_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.graph_cache, + displayLabel="Enable Graph Cache", + inputHelp="Build Graph Cache and wait until it is usable before search", + inputType=InputType.Bool, + inputConfig={"value": True}, +) + +CaseConfigParamInput_GraphCacheTimeout_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.graph_cache_timeout, + displayLabel="Graph Cache timeout (seconds)", + inputHelp="Maximum time to wait for Graph Cache to become usable", + inputType=InputType.Number, + inputConfig={ + "min": 1, + "max": MAX_STREAMLIT_INT, + "value": DEFAULT_GRAPH_CACHE_TIMEOUT_SECONDS, + }, + isDisplayed=lambda config: config.get(CaseConfigParamType.graph_cache, True), +) + +CaseConfigParamInput_IterativeScan_PolarDBPG = CaseConfigInput( + label=CaseConfigParamType.iterative_scan, + displayLabel="Iterative scan", + inputHelp="Graph Cache requires iterative scan to remain off", + inputType=InputType.Option, + inputConfig={"options": ["off", "strict_order", "relaxed_order"]}, + isDisplayed=lambda config: not config.get(CaseConfigParamType.graph_cache, True), +) + + CaseConfigParamInput_IndexType_AlloyDB = CaseConfigInput( label=CaseConfigParamType.IndexType, inputHelp="Select Index Type", @@ -2554,6 +2645,45 @@ class CaseConfigInput(BaseModel): CaseConfigParamInput_quantized_fetch_limit_PgVector, ] +PolarDBPgLoadingConfig = [ + CaseConfigParamInput_IndexType_PolarDBPG, + CaseConfigParamInput_PostLoadIndex_PolarDBPG, + CaseConfigParamInput_m, + CaseConfigParamInput_EFConstruction_PgVector, + CaseConfigParamInput_QuantizationType_PgVector, + CaseConfigParamInput_TableQuantizationType_PgVector, + CaseConfigParamInput_HnswQuantization_PolarDBPG, + CaseConfigParamInput_PQM_PolarDBPG, + CaseConfigParamInput_TrainSamples_PolarDBPG, + CaseConfigParamInput_QuantizationNbits_PolarDBPG, + CaseConfigParamInput_maintenance_work_mem_PgVector, + CaseConfigParamInput_max_parallel_workers_PgVector, + CaseConfigParamInput_GraphCache_PolarDBPG, + CaseConfigParamInput_GraphCacheTimeout_PolarDBPG, +] + +PolarDBPgPerformanceConfig = [ + CaseConfigParamInput_IndexType_PolarDBPG, + CaseConfigParamInput_PostLoadIndex_PolarDBPG, + CaseConfigParamInput_m, + CaseConfigParamInput_EFConstruction_PgVector, + CaseConfigParamInput_EFSearch_PgVector, + CaseConfigParamInput_QuantizationType_PgVector, + CaseConfigParamInput_TableQuantizationType_PgVector, + CaseConfigParamInput_HnswQuantization_PolarDBPG, + CaseConfigParamInput_PQM_PolarDBPG, + CaseConfigParamInput_TrainSamples_PolarDBPG, + CaseConfigParamInput_QuantizationNbits_PolarDBPG, + CaseConfigParamInput_maintenance_work_mem_PgVector, + CaseConfigParamInput_max_parallel_workers_PgVector, + CaseConfigParamInput_reranking_PgVector, + CaseConfigParamInput_reranking_metric_PgVector, + CaseConfigParamInput_quantized_fetch_limit_PgVector, + CaseConfigParamInput_GraphCache_PolarDBPG, + CaseConfigParamInput_GraphCacheTimeout_PolarDBPG, + CaseConfigParamInput_IterativeScan_PolarDBPG, +] + PgVectoRSLoadingConfig = [ CaseConfigParamInput_IndexType_PgVectoRS, CaseConfigParamInput_m, @@ -3444,6 +3574,10 @@ class FilterType(Enum): CaseLabel.Load: PgVectorLoadingConfig, CaseLabel.Performance: PgVectorPerformanceConfig, }, + DB.PolarDBPG: { + CaseLabel.Load: PolarDBPgLoadingConfig, + CaseLabel.Performance: PolarDBPgPerformanceConfig, + }, DB.PgVectoRS: { CaseLabel.Load: PgVectoRSLoadingConfig, CaseLabel.Performance: PgVectoRSPerformanceConfig, diff --git a/vectordb_bench/frontend/config/styles.py b/vectordb_bench/frontend/config/styles.py index 4b162e9ed..eb9186cfc 100644 --- a/vectordb_bench/frontend/config/styles.py +++ b/vectordb_bench/frontend/config/styles.py @@ -48,6 +48,7 @@ def getPatternShape(i): DB.QdrantLocal: "https://assets.zilliz.com/qdrant_b691674fcd.png", DB.WeaviateCloud: "https://assets.zilliz.com/weaviate_4f6f171ebe.png", DB.PgVector: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", + DB.PolarDBPG: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", DB.PgVectoRS: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", DB.PgVectorScale: "data:image/jpeg;base64,/9j/4AAQSkZJRgABAQAAAQABAAD/2wCEAAkGBxAQEhUQEBAWEBAQEBUXEBIVFxgRERASGhUYFhkXGBgaHiggHR0nHRYVLTEhJSkrLjouFx8/OTMsNyg5MS0BCgoKDg0OGBAQGC0eHx8wLS0tLS03KzEtNysrLS0tLSs3Ky03LSstLS0tKyswNi0rNy0tMC4tLSstLS0tLS0tLf/AABEIAMgAyAMBIgACEQEDEQH/xAAcAAEAAgIDAQAAAAAAAAAAAAAABwgBBQMEBgL/xABFEAABAwICBgUJAg0EAwAAAAABAAIDBBEFIQYHEjFBUSIyYXGBExQjQlJikaGxCMEVJDM0Q3J0gpKTorLCFlVj0VNz0v/EABoBAQACAwEAAAAAAAAAAAAAAAACBQEDBgT/xAAqEQACAgIBAwIGAgMAAAAAAAAAAQIDBBExBRIhE0EiMlFSYfCBkTShsf/aAAwDAQACEQMRAD8Ag1ERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAWVutHtF62vds0sDpBexfbZjb3vOQUn4DqN3Oraq3OOAf5u/6Wi3Jrq+ZjRCqyrR4XqzwinGVI2V3tSkyk+By+S9FS4PTRC0VPFGPcja36BeKXVa18sdku0p3snksWVz/ACTfZHwXXqMLp5BaSnikHJzGu+oUF1aP2/7HaU4WFajE9XGE1A6VGxh9qK8JH8OS8Pjuo1hBdRVRaeEcw2m/xt3fAr0V9Spl4fgx2kIIvQaR6G1+Hk+cwOay+UrenEf3hl8V59e6MlJbT2YMIiLICIiAIiIAiIgCIiAIiIAiytjgWDz1szaenZtyPPg0cXOPADmjaS2wdWio5J3tiiY6SR5s1jRtOcewKa9B9TjGBs2J9N+8U7T0G/8AscOt3DLvXstAtBKfC48gJapw9LORn+qz2W/XivXqjyuot/DX4X1JKJwUtNHE0RxMbGxos1jQGtaOwBc6Ljlka0FziGtaCXOOQaBmSTyVX5kyR9oo7w3W3QTVjqU3jivsw1Dj0JHdo9Ucj8bKQwVsspnX8y0Nn0iItICIiA45YmvBa5oc1wsWkXaR2hRdptqfgnDpsPtTzb/JH8g89nFh+XcpUWVvpyJ1PcWOSnOK4XPSyOgqI3RSs3tcLeI5jtXSVsNM9EKbE4vJzN2ZGj0UwHTjP3jsVadKtG6jDpzT1DbHex46krODmldBi5cb19GQa0aREReswEREAREQBERAEREB2KKkfM9sUTS+SRwaxo3uccgFZzVzoVHhcFiA6qlAM8n+DfdHzXi9Reh4a04nM3pOu2lBHVbudJ45gdx5qY1SdRytv048LklFGURFUEgow18Y66CjZTMdsuq3u27bzEyxcPElnwKk9Qr9oqld+KTeoPKsJ5OOy4fEA/BezAinfFMw+CFlKWrPWg+j2aWuJfS7o5Os+Abre8zs3j5KLUXRW1Rtj2yRAuBhePUdV+b1MU3Yx7XO+G9bJUva4g3GRG48l6LCtOsUpreSrZbDc158q23c+4VVZ0r7Jf2S7i1yKAcK131rMqininHEtvC8+OY+S9vgeuPDqhzY5WSUznkDafsuiBPN4N7dpC8dmBdD22Z2iSEXy0g5jMHcea+l42tGTC89ptorDidOYJOi8XMMts4n8+7dcL0KKUJuElKPIKd4zhctJM+nnbsSxOs4fQjsIzXRVg9dmiAqqfz6Fvp6ZvpAN8kG8+Lb37rqvi6jGvV0FL3INaMIiLeYCIiAIiIAtxongrq6rhpW39LIA4j1Wb3u8BdahTL9nvBbunrXDqAQxHtPSf8ALY/iK05FvpVuRlEz0dKyGNsUbQ2ONgaxo4NAsFzrCyuVb29smERFEBaPTHR6LEaV9NLlfpRvGZjkG5w+fgSvJ6wNaUWHytp6djaiZrh5xc2bG32Lj1/ovX6P6QU+IU4qKd+00izmnrRvtm1w5r1Km2pRs1oFWMfwWaimdBMLOaciOq9vBwPJaxTTrKoWTAhwzabtdxaVDM0ZaS07wV0OPd6kE/cgzjREW8wEREBLeqjWV5vs0Nc/0GQgmcfyPJjj7Hbw7t07tcDmMwRkeapcpa1U6y/Ny2hrn3gOUMzj+R4Bjvc7eHduqs7B7tzgvJJMnhFhrgcxmCMjzWQqNokfMjA4EOAIIsQcwQqq6w9HvwfXSwAeiJ24O2J2YHhmP3Va1RH9oLBQ+CGsaOlC/wAm8/8AG/MX7nD+tWPTruy3t9mYkQOiIugIBERAEREBlWf1Q4cKfC4PamDpXfvnL+kNVYArg6OQCOkp4x6lNE34MAVZ1SWq0vqyUTZIiKhJGFptL4qt9JM2hcGVRZ6Mn5hp4Otex5rQaR6zqGiqmUj7yEutPIw3bT8BfmeYG75L2kEzZGh7HB7HAFrmm7XA5gg8lu9OdTjOSBTisikY9zZQ5sgcRIHdYO43vne63Wh2ldRhk3lYjdjspoj1JWfceRU260NXbMQaammAZWsbnwbUAeq73uR8D2V2qIHxucx7Sx7HEOaRZzXDIghdDRdDJh/1EGtEuY9jMNZD5eF12uGYPWY7i13aolruu7vXLQYg+EnZPRdk5vBwXXqXhzi4biVsqq9PwuAcSIi3GAiIgCIiAlvVRrLNPs0Nc+8BygmdmYeAa73Pp3bp3a4EXGYIyPNUuU56h9JqicSUMrttkEYfC45uY2+zsd2eXJVGfhrTsj/JJP2JfXntYWHipw6qiIufIOc39ZnpG/NoXoV8TxB7XMO5zS09xFlUVS7Zp/Qkyl6Lmqo9h7m+y4j4Gy4V1yNYREQBERAZCuXRfk2W3bDbfAKmgVwtH5xJSwSD16eJ3xYCqnqq+GP8komxXjtaeKVtLQvkomXde0sgzdBFY3e0ffw3r0Nbi8EEsMEsgZJUl4gB9ctsSL7r9Id67zm3FjmCMxwKqa36coyktokUwe8uJJNyTck5klSLqt1jOw9wpakl1E85He6nceI93mPEdux1r6tvNtqtom/i5N5oR+hPtN9z6d26J10i9PJr/BDgudBM2Roexwex7Q5rmm7XA53BUf60NXTcRaammAZWsbmNzaho9V3vcj4d0c6rtYzqBwpqkl1G45He6nJ4j3eY8e+w0EzXtD2ODmOALXNN2uBzBBVJZXZiWbXBLkptUwPjc5j2lj2OIc1ws5pG8ELiVjtaGrtuINNTTAMrWNz4NqGgdU+9yPh3V2nhdG4se0te1xDmkWLXDIg9qvMbJjfHa5ItaOFERbzAREQBERAZUxfZ4oJPK1NRb0QjbGDzeXB1h3AfMKJaCkfPIyGNpdJI8NYBxcTYK2Gh+Asw+kipWWuxt5HD15Dm53x+QC8HUblCrt92ZibtYWVxyyBrS47mgk9wzXPR5RMp7jP5xNbd5eS38ZXRXNWS7b3O9p7j8TdcK69cGsIiLICIiAyrR6psRE+F05vcxNMTuwsJA/p2fiquKa/s9YzlUUTjuImiH9D/APD5rwdRr7qdr2Mrk+vtFOI8xINiHVFjxB9Cu/qo1lCpDaKtfaoAtDKchP7rj7f1710ftFxOLaJ1jstdUBzuAJEVh/S74KE2PINwbEG4O4gqFFEbsWMX++TLemXDxcXjIOYIz7VW7T7RlsEjpYBaNxu5g9Q9nYve6vtZPnUYoq11qgC0MpyE43bLvf8Ar3ro6ZDetGNCzHscWH5IcUk6rNYr6Bwpakl1G92R3mnceI93mPHvjysaA8gbrrhCtbK42x7ZIiXRa4EXGYIyPNV7184M2CtZUMGyKuK77cZGHZcfEFinbAPzWD9ni/sCiL7RvWov1Z/rEqPAbjkdq/JN8EMLKwt/SaHYlMxskVDO+N4u1wjdsuHMdiv3JR5ZA0CL0f8AoTFv9vn/AJbk/wBCYt/t8/8ALco+rD7kDzqLb4nozXUzduopJYWX6zmODfiuHAcJkrKiOmiF3zPDRyA4uPYBc+Cl3x1vfgEoahtFdt7sSlb0Y7sp7je/13juBt+8eSnFdHBMLjo4I6aIWjhYGt5nm49pNz4rvLl8q/1rHI2LwFodPMQFNh1VKTa1O9rf1njYb83Bb9RR9oDGRHSxUjT0p5Nt4/42c/3iP4Uxa++2KDICREXUmsIiIAiIgMre6FY6aCshqR1WPtKB60RyePgStCsrEoqSafuC4GKYdTYhTmKVolgmYC09hF2vaeBzyKrXp9oVPhU2y676d5PkJrZOHsu5OHJSfqN0uE0P4Pmd6WAXgJ9eHi3vb9D2KSMbwiCthdT1DA+N4zHEHg5p4Ec1RV2zxLXCXH75J8lPmOINwbEHI7rL19HpMZ4/JTm8rRZrz+kHb2rh0+0KmwqbZdd9O8nyE1snD2XcnDkvKAq5+C2KkiPB2K/rldcLL3k5nMrAW1cGC4eAfmtP+zRf2BRH9o3rUXdP9YlLmAfmtP8As0X9gUR/aO61F3T/AFiXP4X+T/ZN8EMBXJwwAQxACwETLdnRCpsFcXBJ2yU8MjCHNdCwtI3EbIXq6r8sTETvIiKk2SPiRgcCHAOaRYg5ghanDtF6GnlNRBSxxTOBBe0Wy42G4eC3CKSnJLSYMoiKIPlxAzOXNVZ1maQ/hCvllabxRnycHLYbx8TtHxUw659LhR0ppYnfjNU0g23xw7nO8dw8eSrmrvpmPpOx+/BGRhERWxEIiIAiIgCIiA7uFYjLTSsnhdsSxODmOHP7x2K0WgulsOKU4mZZsrbCeLjG/wD+TwKqitxovpFUYdO2op3WcMntPUkZxa4cl5MvFV8fyjKei1eOYRBWwup6hgfG8ZjiDwc08COarPp9oVPhU2y676d5PkJrZOHsu5OHJWF0M0vpsUi8pC7ZkaB5aEnpxn728itnjeEQVsLqeoYHxvGYO8Hg5p4Ec1U4+RPGn2z4+hJrZTxAvV6faFT4VNsuu+neT5Ga2Th7LuThyXk1fxnGcVKL8EC4GjEofR0z27nUsRH8sKPdfeATVEENTE0vbSmTyoGZDHhvT7hs5966WpTTlhjbhlS7Ze0nzV53PaTfyffcm3fZTC5oIIIuCMxzXPS7sW/uaJ8opapJ1W6xnYe4UtSS6iecjvdTuPrD3eY8R27HWvq1832q6hZenJvNCP0HvN9z6d26Jldp15Nf4I8FzoJmyND2OD2PaC1wN2uBzBBXKq7ardYzqBwpapxdRvd0Xb3UxJ3j3eY8R22Fhma9oexwc1wBa4G7XA5ggrn8rGlTLT4Jp7OREReUGFpdLdI4MNp3VEx3ZRsHWlfwaP8AvgmlOk1NhsJmqH2/8cY/KSu9lo+/gq0aZ6WVGKTmaY2Y24iiB6ETOQ5nmeKsMPDdr7n8phvR0dIcamrp31M7rySG/Y1vBo5ALWIsLokklpEAiIgCIiAIiIAiIgCIiA72EYpPSStnp5DFKw5Ob9DzHYVOug+t2nqQ2Gu2aafcJN0Eh7/UPfl2qvqLRfjQuWpIynouDjGFU9dA6Cdolhlb32yyc08+1Vp0+0KmwqbZdd9O8nyE1snD2XcnDkuHRnTnEMPsKecmMb4X+kiPgd3hZSGzWxQV8BpcVpHNbILOfH6RoPtgHpNI7Lrx003Y0vHxRM7TIZY8ggg2INwRvCnvVRrKFUG0Va8CoAAhlOQnGXRd7/171CuP0UMMpFNUNqYHZxSAFrtnk9rgC1wWuY8g3BsQciN4K9l9EL4aZhPRc9zQQQRcEWIO4hQPrX1amnLq2hZenNzNCP0B4ub7n07l6TVTrKFUG0Va+1SMoZTkJxu2Xe/9e9SlKW2O0Rs2zvut2qkg7cS3X6yXJTBSXqr1jOoXCkqnF1G49F291O48R7nMePfw62NGKKmk84oamEskd6SmbI1z4nHO7Wg9Ts4d26OldtQyK/K8MjwXOima5oe1wcxzbtcDdpbvuCo9021r0lGHRUpFXU2t0TeGM7uk4b+4fEKCTpHW+bik85kFM29og6zc87do7N29aleKrpkYy3N7M9xs8fxyorpTPUyGSQ7r9Vo9lo3Adi1iLCtEklpEQiIgCIiAIiIAiIgCIiAIiIAiIgCIiAIiID7Y4g3BsRuI3hck1VI/rvc7vJcuBE0gEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREAREQBERAEREB//2Q==", DB.PgDiskANN: "https://assets.zilliz.com/PG_Vector_d464f2ef5f.png", @@ -88,6 +89,7 @@ def getPatternShape(i): DB.QdrantCloud.value: "#D91AD9", DB.WeaviateCloud.value: "#20C997", DB.PgVector.value: "#4C779A", + DB.PolarDBPG.value: "#2F6FA5", DB.Redis.value: "#0D6EFD", DB.AWSOpenSearch.value: "#0DCAF0", DB.OSSOpenSearch.value: "#0DCAF0", diff --git a/vectordb_bench/models.py b/vectordb_bench/models.py index 21eea9e95..81f0206be 100644 --- a/vectordb_bench/models.py +++ b/vectordb_bench/models.py @@ -177,6 +177,14 @@ class CaseConfigParamType(Enum): post_load_index = "post_load_index" pq_nbits = "pq_nbits" + # PolarDB for PostgreSQL parameters + hnsw_quantization = "hnsw_quantization" + train_samples = "train_samples" + quantization_nbits = "quantization_nbits" + graph_cache = "graph_cache" + graph_cache_timeout = "graph_cache_timeout" + iterative_scan = "iterative_scan" + # Lindorm parameters efSearch = "efSearch" pq_m = "pq_m"