From d58b1a3748e940a1bec0a95785f0b13588430294 Mon Sep 17 00:00:00 2001 From: Owen Halpert Date: Thu, 10 Sep 2026 23:04:49 -0700 Subject: [PATCH 1/4] turbopuffer: consistency level, readiness polling, base64 vectors, base URL fix, README --- README.md | 26 +++++++ .../backend/clients/turbopuffer/cli.py | 11 +++ .../backend/clients/turbopuffer/config.py | 2 + .../clients/turbopuffer/turbopuffer.py | 78 +++++++++++++++---- vectordb_bench/backend/data_source.py | 10 ++- 5 files changed, 110 insertions(+), 17 deletions(-) diff --git a/README.md b/README.md index 3b5635585..5d54649fa 100644 --- a/README.md +++ b/README.md @@ -426,6 +426,32 @@ Options: --help Show this message and exit. ``` +### Run turbopuffer from command line + +```shell +vectordbbench turbopuffer \ + --api-key "$TURBOPUFFER_API_KEY" --region byoc --api-base-url https:// \ + --case-type Performance768D1M \ + --insert-batch-size 10000 --load-concurrency 32 --disable-backpressure \ + --num-concurrency 1,5,10,20,40 +``` + +For the 10M Cohere dataset use `--case-type Performance768D10M`. + +turbopuffer-specific options: +| Option | Description | +|--------|-------------| +| `--region` | turbopuffer region, e.g. `gcp-us-central1`, `aws-us-east-1` (required) | +| `--namespace` | Namespace to write to (default: `vdbbench_test`) | +| `--metric-type` | `COSINE` or `L2` (default: `COSINE`) | +| `--consistency-level` | `strong` or `eventual` (default: `strong`) | +| `--disable-backpressure` | Don't reject writes when indexing falls behind. Recommended for bulk loads; the benchmark waits for indexing to finish before searching anyway | +| `--pin-namespace` / `--pin-replicas` | Pin the namespace to dedicated nodes before the run | + +Each write is one request, so use a large `--insert-batch-size` (10000 is a good start; batches can be up to 512 MB). Set `--load-concurrency` to about 2x the vCPUs of the client; each worker holds one batch in memory (~400 MB at 1536 dims), so lower it on a small box. Run from the same cloud region as the namespace. + +To list all options, run `vectordbbench turbopuffer --help`. + ### Run OceanBase from command line Execute tests for the index types: HNSW, HNSW_SQ, or HNSW_BQ. diff --git a/vectordb_bench/backend/clients/turbopuffer/cli.py b/vectordb_bench/backend/clients/turbopuffer/cli.py index def442d69..b4eb24f6c 100644 --- a/vectordb_bench/backend/clients/turbopuffer/cli.py +++ b/vectordb_bench/backend/clients/turbopuffer/cli.py @@ -110,6 +110,16 @@ class TurboPufferTypedDict(TypedDict): help="Disable Turbopuffer write backpressure", ), ] + consistency_level: Annotated[ + str, + click.option( + "--consistency-level", + type=click.Choice(["strong", "eventual"], case_sensitive=False), + help="Query consistency level (strong or eventual)", + default="eventual", + show_default=True, + ), + ] pin_namespace: Annotated[ bool, click.option( @@ -221,6 +231,7 @@ def TurboPuffer(**parameters: Unpack[TurboPufferIndexTypedDict]): pin_replicas=parameters["pin_replicas"], pin_timeout=parameters["pin_timeout"], pin_target_namespace_count=pin_target_namespace_count, + consistency_level=parameters["consistency_level"].lower(), ), db_case_config=TurboPufferIndexConfig( metric_type=MetricType(parameters["metric_type"]), diff --git a/vectordb_bench/backend/clients/turbopuffer/config.py b/vectordb_bench/backend/clients/turbopuffer/config.py index cdb7e8328..1b8418cf8 100644 --- a/vectordb_bench/backend/clients/turbopuffer/config.py +++ b/vectordb_bench/backend/clients/turbopuffer/config.py @@ -22,6 +22,7 @@ class TurboPufferConfig(DBConfig): pin_replicas: int = 1 pin_timeout: int = 45 * 60 pin_target_namespace_count: int = 0 + consistency_level: str = "eventual" def to_dict(self) -> dict: return { @@ -36,6 +37,7 @@ def to_dict(self) -> dict: "pin_replicas": self.pin_replicas, "pin_timeout": self.pin_timeout, "pin_target_namespace_count": self.pin_target_namespace_count, + "consistency_level": self.consistency_level, } diff --git a/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py b/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py index 0b87462d4..3d83bd363 100644 --- a/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py +++ b/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py @@ -1,5 +1,6 @@ """Wrapper around the TurboPuffer vector database over VectorDB""" +import base64 import logging import os import time @@ -10,6 +11,7 @@ from urllib.parse import quote from urllib.request import Request, urlopen +import numpy as np import turbopuffer as tpuf from vectordb_bench.backend.clients.turbopuffer.config import ( @@ -102,6 +104,7 @@ def __init__( self.region = db_config.get("region", "") self.api_base_url = db_config.get("api_base_url") self.namespace = db_config.get("namespace", "") + self.consistency_level = db_config.get("consistency_level", "eventual") self.multitenant_namespace_prefix = db_config.get("multitenant_namespace_prefix", "vdbbench_mt_") self.multitenant_tenant_labels: list[str] = kwargs.get("multitenant_tenant_labels", []) self._multitenant_touched_tenants: set[str] = set() @@ -142,12 +145,14 @@ def __init__( tmp_client = None def _create_client(self) -> tpuf.Turbopuffer: - client_kwargs = {"api_key": self.api_key, "region": self.region} + client_kwargs = {"api_key": self.api_key} + if self.api_base_url: + client_kwargs["base_url"] = self.api_base_url + else: + client_kwargs["region"] = self.region max_retries = os.getenv("TURBOPUFFER_MAX_RETRIES") if max_retries is not None: client_kwargs["max_retries"] = int(max_retries) - if self.api_base_url: - client_kwargs["base_url"] = self.api_base_url return tpuf.Turbopuffer(**client_kwargs) def _apply_namespace_pinning(self): @@ -211,6 +216,15 @@ def init(self): self._ns_cache = {} self.ns = self.client.namespace(self.namespace) self._apply_namespace_pinning() + # Only warm cache if namespace has data (skip during initial load) + if not self.multitenant_tenant_labels: + try: + row_count = getattr(self.ns.metadata(), "approx_row_count", 0) or 0 + if row_count > 0: + log.debug(f"Namespace has {row_count} rows, ensuring cache is warm...") + self._warm_cache(self.ns) + except Exception as e: + log.warning(f"Could not check namespace metadata: {e}") yield def supports_multitenant(self) -> bool: @@ -233,18 +247,43 @@ def _namespace_for_tenant(self, tenant: str | None): return ns def optimize(self, data_size: int | None = None): - # turbopuffer responds to the request - # once the cache warming operation has been started. - # It does not wait for the operation to complete, - # which can take multiple minutes for large namespaces. warmed_namespaces = self._warmup_target_namespaces() - for namespace in warmed_namespaces: - self._namespace_for_tenant(namespace).hint_cache_warm() if not warmed_namespaces: log.info("TurboPuffer cache warmup skipped") return - log.info(f"warming up but no api waiting for complete. just sleep {self.db_case_config.time_wait_warmup}s") - time.sleep(self.db_case_config.time_wait_warmup) + for namespace in warmed_namespaces: + ns = self._namespace_for_tenant(namespace) + self._wait_for_index(ns) + self._warm_cache(ns) + + @staticmethod + def _wait_for_index(ns: Any): + """Wait for index to be fully built.""" + while True: + index = ns.metadata().index + status = getattr(index, "status", None) + if status == "up-to-date": + log.info("Index is up-to-date") + return + unindexed = getattr(index, "unindexed_bytes", None) + log.info(f"Index status: {status}, unindexed_bytes: {unindexed}. Checking again in 10s...") + time.sleep(10) + + @staticmethod + def _warm_cache(ns: Any): + """Start cache warming and poll until complete.""" + # First call returns "cache warm hint accepted" + # Subsequent calls while warming return "cache is already warming" + # When warming is done, calling again returns "cache warm hint accepted" + log.debug("Starting cache warm...") + ns.hint_cache_warm() + while True: + time.sleep(5) + response = ns.hint_cache_warm() + log.debug(f"Cache warm response: {response}") + if "accepted" in str(response).lower(): + log.debug("Cache warming complete") + return def _warmup_target_namespaces(self) -> list[str | None]: if not self.multitenant_tenant_labels: @@ -258,6 +297,16 @@ def _warmup_target_namespaces(self) -> list[str | None]: return self.multitenant_tenant_labels return [] + @staticmethod + def _encode_vector(embedding: list[float] | np.ndarray) -> str: + arr = np.ascontiguousarray(np.asarray(embedding, dtype=" list[str]: + arr = np.ascontiguousarray(np.asarray(embeddings, dtype=" tuple[int, Exception]: - vectors = [embedding.tolist() if hasattr(embedding, "tolist") else embedding for embedding in embeddings] + vectors = self._encode_vectors(embeddings) if tenant_labels_data is not None: inserted = 0 successful_tenants: dict[str, int] = {} @@ -432,12 +481,15 @@ def search_embedding( tenant: str | None = None, ) -> list[int]: query_kwargs = { - "rank_by": ("vector", "ANN", query), + "rank_by": ("vector", "ANN", self._encode_vector(query)), "top_k": k, "filters": self.expr, } + if self.consistency_level == "eventual": + query_kwargs["consistency"] = {"level": "eventual"} if payload_profile == PayloadProfile.VECTOR: query_kwargs["include_attributes"] = [self._vector_field] + query_kwargs["vector_encoding"] = "base64" elif payload_profile == PayloadProfile.SCALAR_LABEL: query_kwargs["include_attributes"] = [self._scalar_payload_label_field] res = self._namespace_for_tenant(tenant).query(**query_kwargs) diff --git a/vectordb_bench/backend/data_source.py b/vectordb_bench/backend/data_source.py index 07ac8414a..b7ac2cf11 100644 --- a/vectordb_bench/backend/data_source.py +++ b/vectordb_bench/backend/data_source.py @@ -4,6 +4,7 @@ import pathlib import typing from abc import ABC, abstractmethod +from concurrent.futures import ThreadPoolExecutor, as_completed from enum import Enum import ir_datasets @@ -178,10 +179,11 @@ def read( if len(downloads) == 0: return {file: local_ds_root / file for file in files} - log.info(f"Start to downloading files, total count: {len(downloads)}") - for s3_file in tqdm(downloads): - log.debug(f"downloading file {s3_file} to {local_ds_root}") - self.fs.download(s3_file, local_ds_root.as_posix()) + log.info(f"Start to downloading files in parallel, total count: {len(downloads)}") + with ThreadPoolExecutor(max_workers=12) as pool: + futures = [pool.submit(self.fs.download, s3_file, local_ds_root.as_posix()) for s3_file in downloads] + for future in tqdm(as_completed(futures), total=len(futures), desc="Downloading"): + future.result() log.info(f"Succeed to download all files, downloaded file count = {len(downloads)}") return {file: local_ds_root / file for file in files} From 10c353dede5481ac4267bd5869ce26d63648a677 Mon Sep 17 00:00:00 2001 From: Owen Halpert Date: Fri, 18 Sep 2026 17:53:01 -0700 Subject: [PATCH 2/4] turbopuffer: add --probes and --vector-type, use raw query responses --- README.md | 5 +++- .../backend/clients/turbopuffer/cli.py | 24 +++++++++++++++++++ .../backend/clients/turbopuffer/config.py | 3 +++ .../clients/turbopuffer/turbopuffer.py | 19 +++++++++++++-- 4 files changed, 48 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index 5d54649fa..1c2faa76d 100644 --- a/README.md +++ b/README.md @@ -433,7 +433,8 @@ vectordbbench turbopuffer \ --api-key "$TURBOPUFFER_API_KEY" --region byoc --api-base-url https:// \ --case-type Performance768D1M \ --insert-batch-size 10000 --load-concurrency 32 --disable-backpressure \ - --num-concurrency 1,5,10,20,40 + --probes 400 --vector-type f16 \ + --num-concurrency 10,40,80,160 ``` For the 10M Cohere dataset use `--case-type Performance768D10M`. @@ -445,6 +446,8 @@ turbopuffer-specific options: | `--namespace` | Namespace to write to (default: `vdbbench_test`) | | `--metric-type` | `COSINE` or `L2` (default: `COSINE`) | | `--consistency-level` | `strong` or `eventual` (default: `strong`) | +| `--probes` | ANN probes per query. Unset uses the server default (a fixed 200) | +| `--vector-type` | Stored vector element type, `f32` or `f16` (default: `f32`). `f16` halves storage; the wire format stays `f32` | | `--disable-backpressure` | Don't reject writes when indexing falls behind. Recommended for bulk loads; the benchmark waits for indexing to finish before searching anyway | | `--pin-namespace` / `--pin-replicas` | Pin the namespace to dedicated nodes before the run | diff --git a/vectordb_bench/backend/clients/turbopuffer/cli.py b/vectordb_bench/backend/clients/turbopuffer/cli.py index b4eb24f6c..61ed732e2 100644 --- a/vectordb_bench/backend/clients/turbopuffer/cli.py +++ b/vectordb_bench/backend/clients/turbopuffer/cli.py @@ -140,6 +140,28 @@ class TurboPufferTypedDict(TypedDict): ), ] pin_timeout: PinTimeoutOption + probes: Annotated[ + int | None, + click.option( + "--probes", + type=click.IntRange(min=1), + default=None, + help=( + "ANN probes per query. Server default is a fixed 200 (~0.90 recall@100 on " + "Cohere 1M); 400 gives ~0.95 and 800 ~0.98, trading throughput for recall." + ), + ), + ] + vector_type: Annotated[ + str, + click.option( + "--vector-type", + type=click.Choice(["f32", "f16"], case_sensitive=False), + default="f32", + show_default=True, + help="Stored vector element type. f16 halves storage; the wire format stays f32.", + ), + ] multitenant_warmup_policy: Annotated[ str, click.option( @@ -235,6 +257,8 @@ def TurboPuffer(**parameters: Unpack[TurboPufferIndexTypedDict]): ), db_case_config=TurboPufferIndexConfig( metric_type=MetricType(parameters["metric_type"]), + probes=parameters["probes"], + vector_type=parameters["vector_type"].lower(), disable_backpressure=parameters["disable_backpressure"], multitenant_warmup_policy=TurboPufferMultitenantWarmupPolicy(parameters["multitenant_warmup_policy"]), ), diff --git a/vectordb_bench/backend/clients/turbopuffer/config.py b/vectordb_bench/backend/clients/turbopuffer/config.py index 1b8418cf8..988c7062e 100644 --- a/vectordb_bench/backend/clients/turbopuffer/config.py +++ b/vectordb_bench/backend/clients/turbopuffer/config.py @@ -1,4 +1,5 @@ from enum import StrEnum +from typing import Literal from pydantic import BaseModel, SecretStr @@ -43,6 +44,8 @@ def to_dict(self) -> dict: class TurboPufferIndexConfig(BaseModel, DBCaseConfig): metric_type: MetricType | None = None + probes: int | None = None + vector_type: Literal["f32", "f16"] = "f32" use_multi_ns_for_filter: bool = False time_wait_warmup: int = 60 * 1 # 1min disable_backpressure: bool = False diff --git a/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py b/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py index 3d83bd363..2d9c8e497 100644 --- a/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py +++ b/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py @@ -12,6 +12,7 @@ from urllib.request import Request, urlopen import numpy as np +import orjson import turbopuffer as tpuf from vectordb_bench.backend.clients.turbopuffer.config import ( @@ -114,6 +115,7 @@ def __init__( self.pin_timeout = db_config.get("pin_timeout", PINNING_TIMEOUT) self._pinning_applied = False self.db_case_config = db_case_config + self.dim = dim self._is_fts = isinstance(db_case_config, TurboPufferFtsConfig) self.metric = None if self._is_fts else db_case_config.parse_metric() @@ -297,6 +299,11 @@ def _warmup_target_namespaces(self) -> list[str | None]: return self.multitenant_tenant_labels return [] + def _vector_schema(self) -> dict | None: + if self.db_case_config.vector_type == "f32": + return None + return {self._vector_field: {"type": f"[{self.dim}]{self.db_case_config.vector_type}", "ann": True}} + @staticmethod def _encode_vector(embedding: list[float] | np.ndarray) -> str: arr = np.ascontiguousarray(np.asarray(embedding, dtype=" Date: Mon, 21 Sep 2026 21:02:22 -0400 Subject: [PATCH 3/4] turbopuffer: hint_read_only --- tests/test_turbopuffer_cli.py | 30 +++++++++- .../clients/turbopuffer/turbopuffer.py | 60 +++++++++++++++++-- 2 files changed, 85 insertions(+), 5 deletions(-) diff --git a/tests/test_turbopuffer_cli.py b/tests/test_turbopuffer_cli.py index 46db473ca..49ab8e575 100644 --- a/tests/test_turbopuffer_cli.py +++ b/tests/test_turbopuffer_cli.py @@ -141,6 +141,7 @@ def test_turbopuffer_multitenant_optimize_skips_base_namespace_by_default( monkeypatch: MonkeyPatch, ) -> None: warmed = [] + hinted = [] class FakeNamespace: def __init__(self, name: str): @@ -148,16 +149,28 @@ def __init__(self, name: str): def hint_cache_warm(self): warmed.append(self.name) + return "cache warm hint accepted" + + def metadata(self): + return SimpleNamespace(index=SimpleNamespace(status="up-to-date", unindexed_bytes=0)) class FakeClient: def namespace(self, name: str): return FakeNamespace(name) + def fake_hint_read_only(api_key, region, namespace, api_base_url=None): + hinted.append(namespace) + return {"status": "accepted"} + monkeypatch.setattr(turbopuffer_client.time, "sleep", lambda _seconds: None) + monkeypatch.setattr(turbopuffer_client, "namespace_hint_read_only_request", fake_hint_read_only) db = object.__new__(TurboPuffer) db.client = FakeClient() db.ns = FakeNamespace("base") db.namespace = "base" + db.api_key = "secret" + db.region = "aws-us-west-2" + db.api_base_url = None db.multitenant_namespace_prefix = "mt_" db.multitenant_tenant_labels = ["tenant_0000", "tenant_0001"] db._ns_cache = {} @@ -166,12 +179,14 @@ def namespace(self, name: str): db.optimize() assert warmed == [] + assert hinted == ["mt_tenant_0000", "mt_tenant_0001"] def test_turbopuffer_multitenant_optimize_can_warm_all_tenant_namespaces( monkeypatch: MonkeyPatch, ) -> None: warmed = [] + hinted = [] class FakeNamespace: def __init__(self, name: str): @@ -179,16 +194,28 @@ def __init__(self, name: str): def hint_cache_warm(self): warmed.append(self.name) + return "cache warm hint accepted" + + def metadata(self): + return SimpleNamespace(index=SimpleNamespace(status="up-to-date", unindexed_bytes=0)) class FakeClient: def namespace(self, name: str): return FakeNamespace(name) + def fake_hint_read_only(api_key, region, namespace, api_base_url=None): + hinted.append(namespace) + return {"status": "accepted"} + monkeypatch.setattr(turbopuffer_client.time, "sleep", lambda _seconds: None) + monkeypatch.setattr(turbopuffer_client, "namespace_hint_read_only_request", fake_hint_read_only) db = object.__new__(TurboPuffer) db.client = FakeClient() db.ns = FakeNamespace("base") db.namespace = "base" + db.api_key = "secret" + db.region = "aws-us-west-2" + db.api_base_url = None db.multitenant_namespace_prefix = "mt_" db.multitenant_tenant_labels = ["tenant_0000", "tenant_0001"] db._ns_cache = {} @@ -196,7 +223,8 @@ def namespace(self, name: str): db.optimize() - assert warmed == ["mt_tenant_0000", "mt_tenant_0001"] + assert warmed == ["mt_tenant_0000", "mt_tenant_0000", "mt_tenant_0001", "mt_tenant_0001"] + assert hinted == ["mt_tenant_0000", "mt_tenant_0001"] def test_turbopuffer_cli_skips_multitenant_pin_namespaces_during_dry_run( diff --git a/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py b/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py index 2d9c8e497..a1b4afeed 100644 --- a/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py +++ b/vectordb_bench/backend/clients/turbopuffer/turbopuffer.py @@ -59,6 +59,33 @@ def namespace_metadata_request( raise RuntimeError(msg) from e +def namespace_hint_read_only_request( + api_key: str, + region: str, + namespace: str, + api_base_url: str | None = None, +) -> dict: + """Hint that a namespace's write workload is done, triggering a full LSM compaction.""" + base_url = api_base_url or f"https://{region}.turbopuffer.com" + url = f"{base_url.rstrip('/')}/v2/namespaces/{quote(namespace, safe='')}/hint_read_only" + req = Request( # noqa: S310 + url, + data=b"", + method="POST", + headers={ + "Authorization": f"Bearer {api_key}", + "Content-Type": "application/json", + }, + ) + try: + with urlopen(req, timeout=60) as resp: # noqa: S310 + return loads(resp.read().decode() or "{}") + except HTTPError as e: + detail = e.read().decode(errors="replace") + msg = f"Failed to send TurboPuffer hint_read_only for namespace {namespace}: {e.code} {detail}" + raise RuntimeError(msg) from e + + def wait_for_namespace_pinning( api_key: str, region: str, @@ -249,6 +276,10 @@ def _namespace_for_tenant(self, tenant: str | None): return ns def optimize(self, data_size: int | None = None): + hint_targets = self._hint_read_only_targets() + self._hint_read_only(hint_targets) + for tenant in hint_targets: + self._wait_for_index(self._namespace_for_tenant(tenant)) warmed_namespaces = self._warmup_target_namespaces() if not warmed_namespaces: log.info("TurboPuffer cache warmup skipped") @@ -258,6 +289,30 @@ def optimize(self, data_size: int | None = None): self._wait_for_index(ns) self._warm_cache(ns) + def _touched_tenant_labels(self) -> list[str]: + return sorted(getattr(self, "_multitenant_touched_tenants", set())) or self.multitenant_tenant_labels + + def _hint_read_only_targets(self) -> list[str | None]: + if self.multitenant_tenant_labels: + return self._touched_tenant_labels() + return [None] + + def _hint_read_only(self, targets: list[str | None]) -> None: + """Hint each namespace's write workload is done and trigger a full LSM + compaction. + + optimize() waits for index.status to report up-to-date afterward, which + is a reasonable completion proxy given the typical load-then-optimize + sequencing. + """ + for tenant in targets: + namespace = self._namespace_name_for_tenant(tenant) + try: + response = namespace_hint_read_only_request(self.api_key, self.region, namespace, self.api_base_url) + log.info(f"TurboPuffer hint_read_only for {namespace}: {response}") + except Exception as e: + log.warning(f"Failed to send TurboPuffer hint_read_only for {namespace}. Error: {e}") + @staticmethod def _wait_for_index(ns: Any): """Wait for index to be fully built.""" @@ -400,10 +455,7 @@ def supports_payload_profile(payload_profile: PayloadProfile) -> bool: def poll_insert_readiness(self, expected_count: int) -> dict: if getattr(self, "multitenant_tenant_labels", []): unindexed_by_tenant = {} - tenant_labels = ( - sorted(getattr(self, "_multitenant_touched_tenants", set())) or self.multitenant_tenant_labels - ) - for tenant in tenant_labels: + for tenant in self._touched_tenant_labels(): metadata = self._namespace_for_tenant(tenant).metadata() if not isinstance(metadata, dict): metadata = metadata.model_dump() if hasattr(metadata, "model_dump") else vars(metadata) From b16f6f37831bd167d74d1e02816f29fbfa9b141b Mon Sep 17 00:00:00 2001 From: Nathan VanBenschoten Date: Wed, 23 Sep 2026 11:09:01 -0400 Subject: [PATCH 4/4] turbopuffer: update readme --- README.md | 29 +++++++++++++---------------- 1 file changed, 13 insertions(+), 16 deletions(-) diff --git a/README.md b/README.md index 1c2faa76d..d11ae7b8c 100644 --- a/README.md +++ b/README.md @@ -430,10 +430,10 @@ Options: ```shell vectordbbench turbopuffer \ - --api-key "$TURBOPUFFER_API_KEY" --region byoc --api-base-url https:// \ + --api-key "$TURBOPUFFER_API_KEY" --region "byoc" --api-base-url "https://" \ --case-type Performance768D1M \ --insert-batch-size 10000 --load-concurrency 32 --disable-backpressure \ - --probes 400 --vector-type f16 \ + --probes 200 --vector-type f16 \ --num-concurrency 10,40,80,160 ``` @@ -445,8 +445,8 @@ turbopuffer-specific options: | `--region` | turbopuffer region, e.g. `gcp-us-central1`, `aws-us-east-1` (required) | | `--namespace` | Namespace to write to (default: `vdbbench_test`) | | `--metric-type` | `COSINE` or `L2` (default: `COSINE`) | -| `--consistency-level` | `strong` or `eventual` (default: `strong`) | -| `--probes` | ANN probes per query. Unset uses the server default (a fixed 200) | +| `--consistency-level` | `strong` or `eventual` (default: `eventual`) | +| `--probes` | ANN probes per query. Unset uses the server default (dynamic) | | `--vector-type` | Stored vector element type, `f32` or `f16` (default: `f32`). `f16` halves storage; the wire format stays `f32` | | `--disable-backpressure` | Don't reject writes when indexing falls behind. Recommended for bulk loads; the benchmark waits for indexing to finish before searching anyway | | `--pin-namespace` / `--pin-replicas` | Pin the namespace to dedicated nodes before the run | @@ -1198,20 +1198,17 @@ from vectordb_bench.backend.clients import DB class ZillizTypedDict(CommonTypedDict): - uri: Annotated[ - str, click.option("--uri", type=str, help="uri connection string", required=True) - ] - user_name: Annotated[ - str, click.option("--user-name", type=str, help="Db username", required=True) - ] + uri: Annotated[str, click.option("--uri", type=str, help="uri connection string", required=True)] + user_name: Annotated[str, click.option("--user-name", type=str, help="Db username", required=True)] password: Annotated[ str, - click.option("--password", - type=str, - help="Zilliz password", - default=lambda: os.environ.get("ZILLIZ_PASSWORD", ""), - show_default="$ZILLIZ_PASSWORD", - ), + click.option( + "--password", + type=str, + help="Zilliz password", + default=lambda: os.environ.get("ZILLIZ_PASSWORD", ""), + show_default="$ZILLIZ_PASSWORD", + ), ] level: Annotated[ str,