Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
50 changes: 38 additions & 12 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -426,6 +426,35 @@ Options:
--help Show this message and exit.
```

### Run turbopuffer from command line

```shell
vectordbbench turbopuffer \
--api-key "$TURBOPUFFER_API_KEY" --region "byoc" --api-base-url "https://<cluster endpoint>" \
--case-type Performance768D1M \
--insert-batch-size 10000 --load-concurrency 32 --disable-backpressure \
--probes 200 --vector-type f16 \
--num-concurrency 10,40,80,160
```

For the 10M Cohere dataset use `--case-type Performance768D10M`.

turbopuffer-specific options:
| Option | Description |
|--------|-------------|
| `--region` | turbopuffer region, e.g. `gcp-us-central1`, `aws-us-east-1` (required) |
| `--namespace` | Namespace to write to (default: `vdbbench_test`) |
| `--metric-type` | `COSINE` or `L2` (default: `COSINE`) |
| `--consistency-level` | `strong` or `eventual` (default: `eventual`) |
| `--probes` | ANN probes per query. Unset uses the server default (dynamic) |
| `--vector-type` | Stored vector element type, `f32` or `f16` (default: `f32`). `f16` halves storage; the wire format stays `f32` |
| `--disable-backpressure` | Don't reject writes when indexing falls behind. Recommended for bulk loads; the benchmark waits for indexing to finish before searching anyway |
| `--pin-namespace` / `--pin-replicas` | Pin the namespace to dedicated nodes before the run |

Each write is one request, so use a large `--insert-batch-size` (10000 is a good start; batches can be up to 512 MB). Set `--load-concurrency` to about 2x the vCPUs of the client; each worker holds one batch in memory (~400 MB at 1536 dims), so lower it on a small box. Run from the same cloud region as the namespace.

To list all options, run `vectordbbench turbopuffer --help`.

### Run OceanBase from command line

Execute tests for the index types: HNSW, HNSW_SQ, or HNSW_BQ.
Expand Down Expand Up @@ -1169,20 +1198,17 @@ from vectordb_bench.backend.clients import DB


class ZillizTypedDict(CommonTypedDict):
uri: Annotated[
str, click.option("--uri", type=str, help="uri connection string", required=True)
]
user_name: Annotated[
str, click.option("--user-name", type=str, help="Db username", required=True)
]
uri: Annotated[str, click.option("--uri", type=str, help="uri connection string", required=True)]
user_name: Annotated[str, click.option("--user-name", type=str, help="Db username", required=True)]
password: Annotated[
str,
click.option("--password",
type=str,
help="Zilliz password",
default=lambda: os.environ.get("ZILLIZ_PASSWORD", ""),
show_default="$ZILLIZ_PASSWORD",
),
click.option(
"--password",
type=str,
help="Zilliz password",
default=lambda: os.environ.get("ZILLIZ_PASSWORD", ""),
show_default="$ZILLIZ_PASSWORD",
),
]
level: Annotated[
str,
Expand Down
30 changes: 29 additions & 1 deletion tests/test_turbopuffer_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -141,23 +141,36 @@ def test_turbopuffer_multitenant_optimize_skips_base_namespace_by_default(
monkeypatch: MonkeyPatch,
) -> None:
warmed = []
hinted = []

class FakeNamespace:
def __init__(self, name: str):
self.name = name

def hint_cache_warm(self):
warmed.append(self.name)
return "cache warm hint accepted"

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

tests/test_turbopuffer_cli.py line:152
Medium ---- This fake cannot catch the premature return in _warm_cache: the real SDK returns NamespaceHintCacheWarmResponse(status='ACCEPTED', message=...), so str(response).lower() always contains "accepted" and the poll exits on the first call regardless of warm state. A faithful double that returns a model with status always "ACCEPTED" and a different message while warming is in progress would expose the bug in turbopuffer.py line 341. There is also no test coverage for the new base64 vector encoding, f16 schema generation, consistency param, or probes wiring in the insert/search paths.


def metadata(self):
return SimpleNamespace(index=SimpleNamespace(status="up-to-date", unindexed_bytes=0))

class FakeClient:
def namespace(self, name: str):
return FakeNamespace(name)

def fake_hint_read_only(api_key, region, namespace, api_base_url=None):
hinted.append(namespace)
return {"status": "accepted"}

monkeypatch.setattr(turbopuffer_client.time, "sleep", lambda _seconds: None)
monkeypatch.setattr(turbopuffer_client, "namespace_hint_read_only_request", fake_hint_read_only)
db = object.__new__(TurboPuffer)
db.client = FakeClient()
db.ns = FakeNamespace("base")
db.namespace = "base"
db.api_key = "secret"
db.region = "aws-us-west-2"
db.api_base_url = None
db.multitenant_namespace_prefix = "mt_"
db.multitenant_tenant_labels = ["tenant_0000", "tenant_0001"]
db._ns_cache = {}
Expand All @@ -166,37 +179,52 @@ def namespace(self, name: str):
db.optimize()

assert warmed == []
assert hinted == ["mt_tenant_0000", "mt_tenant_0001"]


def test_turbopuffer_multitenant_optimize_can_warm_all_tenant_namespaces(
monkeypatch: MonkeyPatch,
) -> None:
warmed = []
hinted = []

class FakeNamespace:
def __init__(self, name: str):
self.name = name

def hint_cache_warm(self):
warmed.append(self.name)
return "cache warm hint accepted"

def metadata(self):
return SimpleNamespace(index=SimpleNamespace(status="up-to-date", unindexed_bytes=0))

class FakeClient:
def namespace(self, name: str):
return FakeNamespace(name)

def fake_hint_read_only(api_key, region, namespace, api_base_url=None):
hinted.append(namespace)
return {"status": "accepted"}

monkeypatch.setattr(turbopuffer_client.time, "sleep", lambda _seconds: None)
monkeypatch.setattr(turbopuffer_client, "namespace_hint_read_only_request", fake_hint_read_only)
db = object.__new__(TurboPuffer)
db.client = FakeClient()
db.ns = FakeNamespace("base")
db.namespace = "base"
db.api_key = "secret"
db.region = "aws-us-west-2"
db.api_base_url = None
db.multitenant_namespace_prefix = "mt_"
db.multitenant_tenant_labels = ["tenant_0000", "tenant_0001"]
db._ns_cache = {}
db.db_case_config = SimpleNamespace(time_wait_warmup=1, multitenant_warmup_policy="all")

db.optimize()

assert warmed == ["mt_tenant_0000", "mt_tenant_0001"]
assert warmed == ["mt_tenant_0000", "mt_tenant_0000", "mt_tenant_0001", "mt_tenant_0001"]
assert hinted == ["mt_tenant_0000", "mt_tenant_0001"]


def test_turbopuffer_cli_skips_multitenant_pin_namespaces_during_dry_run(
Expand Down
35 changes: 35 additions & 0 deletions vectordb_bench/backend/clients/turbopuffer/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -110,6 +110,16 @@ class TurboPufferTypedDict(TypedDict):
help="Disable Turbopuffer write backpressure",
),
]
consistency_level: Annotated[
str,
click.option(
"--consistency-level",
type=click.Choice(["strong", "eventual"], case_sensitive=False),
help="Query consistency level (strong or eventual)",
default="eventual",
show_default=True,
),
]
pin_namespace: Annotated[
bool,
click.option(
Expand All @@ -130,6 +140,28 @@ class TurboPufferTypedDict(TypedDict):
),
]
pin_timeout: PinTimeoutOption
probes: Annotated[
int | None,
click.option(
"--probes",
type=click.IntRange(min=1),
default=None,
help=(
"ANN probes per query. Server default is a fixed 200 (~0.90 recall@100 on "
"Cohere 1M); 400 gives ~0.95 and 800 ~0.98, trading throughput for recall."
),
),
]
vector_type: Annotated[
str,
click.option(
"--vector-type",
type=click.Choice(["f32", "f16"], case_sensitive=False),
default="f32",
show_default=True,
help="Stored vector element type. f16 halves storage; the wire format stays f32.",
),
]
multitenant_warmup_policy: Annotated[
str,
click.option(
Expand Down Expand Up @@ -221,9 +253,12 @@ def TurboPuffer(**parameters: Unpack[TurboPufferIndexTypedDict]):
pin_replicas=parameters["pin_replicas"],
pin_timeout=parameters["pin_timeout"],
pin_target_namespace_count=pin_target_namespace_count,
consistency_level=parameters["consistency_level"].lower(),
),
db_case_config=TurboPufferIndexConfig(
metric_type=MetricType(parameters["metric_type"]),
probes=parameters["probes"],
vector_type=parameters["vector_type"].lower(),
disable_backpressure=parameters["disable_backpressure"],
multitenant_warmup_policy=TurboPufferMultitenantWarmupPolicy(parameters["multitenant_warmup_policy"]),
),
Expand Down
5 changes: 5 additions & 0 deletions vectordb_bench/backend/clients/turbopuffer/config.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
from enum import StrEnum
from typing import Literal

from pydantic import BaseModel, SecretStr

Expand All @@ -22,6 +23,7 @@ class TurboPufferConfig(DBConfig):
pin_replicas: int = 1
pin_timeout: int = 45 * 60
pin_target_namespace_count: int = 0
consistency_level: str = "eventual"

def to_dict(self) -> dict:
return {
Expand All @@ -36,11 +38,14 @@ def to_dict(self) -> dict:
"pin_replicas": self.pin_replicas,
"pin_timeout": self.pin_timeout,
"pin_target_namespace_count": self.pin_target_namespace_count,
"consistency_level": self.consistency_level,
}


class TurboPufferIndexConfig(BaseModel, DBCaseConfig):
metric_type: MetricType | None = None
probes: int | None = None
vector_type: Literal["f32", "f16"] = "f32"
use_multi_ns_for_filter: bool = False
time_wait_warmup: int = 60 * 1 # 1min
disable_backpressure: bool = False
Expand Down
Loading
Loading