diff --git a/CONTRIBUTING.rst b/CONTRIBUTING.rst
index 6ce8b6451..f6b820ede 100644
--- a/CONTRIBUTING.rst
+++ b/CONTRIBUTING.rst
@@ -322,6 +322,44 @@ If you prefer to use tox, these flags all work the same way.
tox tests/sql/parsing/queries/tpcds/test_tpcds.py::test_parsing_sparksql_tpcds_queries -- --tpcds
+Property-Based Tests
+--------------------
+
+``datajunction-server/tests/property/`` holds `Hypothesis `_ tests. They generate
+metric definitions, small fact and dimension tables, requests, and SQL expressions to check correctness rules. Most
+execute DJ's SQL on DuckDB and compare it with a direct calculation or query:
+
+- ``metric_decomposition_test.py``: a metric rolled up from its components equals the metric computed directly.
+- ``sql_roundtrip_test.py``: parsing and printing SQL keeps its meaning.
+- ``ast_printing_test.py``: AST-built expressions with unambiguous operator precedence print as equivalent SQL.
+- ``random_graph_test.py``: SQL for a random graph and request matches a plain query over the raw tables.
+- ``materialization_test.py``: a request served from a pre-aggregation matches the same request computed from raw
+ tables. Both paths use DJ-generated SQL; ``random_graph_test.py`` supplies the independent raw-table comparison.
+
+``tests/property/scenario.py`` defines the immutable case, generated metric and filter specifications, and strategies.
+Each metric's raw-table reference expression is written separately from its DJ definition. ``graph.py`` installs a
+case through DJ's API and runs the direct DuckDB reference query. Add new case variants in ``scenario.py`` so the same
+specifications can be reused across properties.
+
+They run with the rest of the server suite. ``DJ_PBT_PROFILE`` selects ``dev`` (the default locally, random with
+shrinking), ``ci`` (the default when ``CI`` is set, a fixed sequence without shrinking), or ``nightly`` (ten times the
+examples, random with shrinking). CI runs only ``ci``, which replays the same examples every time, so it does not
+search for new bugs. ``nightly`` is not scheduled anywhere; run it by hand for a deeper search. The API-backed properties need the same Postgres test setup as the server suite;
+locally, the test fixtures start it through Docker.
+
+.. code-block:: sh
+
+ cd datajunction-server
+ uv run pytest tests/property -n auto
+ DJ_PBT_PROFILE=nightly uv run pytest tests/property -n auto
+
+In ``dev`` and ``nightly``, Hypothesis shrinks failures to a smaller example. The ``ci`` profile reports its failing
+example without shrinking. The ``@reproduce_failure`` decorator in failure output can replay an example exactly.
+
+Bugs found this way and not yet fixed are listed in ``tests/property/known_issues.py``. The generated tests exclude
+affected inputs or metric families, and each listed bug has a small repro marked ``xfail(strict=True)``. Fixing a bug
+makes its repro pass, which fails the run; then remove the marker and its corresponding exclusion.
+
Enabling ``pdb`` When Running Tests
-----------------------------------
@@ -432,4 +470,4 @@ The easiest way to fix it is to reset your database state using these commands (
root@...:/code# alembic upgrade head
...
-After this, the `docker compose up` command should start the db_migration agent without problems.
\ No newline at end of file
+After this, the `docker compose up` command should start the db_migration agent without problems.
diff --git a/datajunction-server/pyproject.toml b/datajunction-server/pyproject.toml
index 45df7edeb..cc2f43f8a 100644
--- a/datajunction-server/pyproject.toml
+++ b/datajunction-server/pyproject.toml
@@ -161,6 +161,7 @@ testpaths = [
]
norecursedirs = [
"tests/helpers",
+ ".hypothesis",
]
[tool.ruff.lint]
@@ -188,4 +189,5 @@ test = [
"sqlparse<1.0.0,>=0.4.3",
"asgi-lifespan>=2",
"mcp>=1.0.0",
+ "hypothesis>=6.168.1",
]
diff --git a/datajunction-server/tests/property/__init__.py b/datajunction-server/tests/property/__init__.py
new file mode 100644
index 000000000..e69de29bb
diff --git a/datajunction-server/tests/property/ast_printing_test.py b/datajunction-server/tests/property/ast_printing_test.py
new file mode 100644
index 000000000..6806437f9
--- /dev/null
+++ b/datajunction-server/tests/property/ast_printing_test.py
@@ -0,0 +1,203 @@
+"""
+Property: an expression tree built in code prints as SQL that means the same
+thing.
+
+DJ assembles SQL by constructing AST nodes (metric combiners, combined filters,
+derived-metric substitution). This property covers expression trees whose
+operators have unambiguous precedence while the known nested-parentheses bug
+has a separate, fixed repro. Each tree is rendered by DJ and by a reference
+renderer that parenthesizes every sub-expression, then evaluated on DuckDB.
+"""
+
+import duckdb
+import pytest
+from hypothesis import given, settings
+from hypothesis import strategies as st
+
+from datajunction_server.sql.parsing import ast
+from tests.property.budget import examples
+from tests.property.comparison import PropertyMismatch
+from tests.property.known_issues import PRINTER_DROPS_PARENTHESES
+
+
+class Expr:
+ """A generated expression: its DJ AST and its fully parenthesized SQL."""
+
+ def __init__(self, build, reference: str):
+ self.build = build # fresh AST on each call; nodes carry parent links
+ self.reference = reference
+
+ def __repr__(self) -> str:
+ return self.reference
+
+
+def column(name):
+ return Expr(lambda: ast.Column(ast.Name(name)), name)
+
+
+def number(value):
+ return Expr(lambda: ast.Number(value), f"({value})" if value < 0 else str(value))
+
+
+NULL = Expr(lambda: ast.Null(), "NULL")
+
+
+def binary(op: ast.BinaryOpKind, left: Expr, right: Expr) -> Expr:
+ return Expr(
+ lambda: ast.BinaryOp(op=op, left=left.build(), right=right.build()),
+ f"({left.reference} {op.value} {right.reference})",
+ )
+
+
+def negate(e: Expr) -> Expr:
+ return Expr(
+ lambda: ast.ArithmeticUnaryOp(
+ op=ast.ArithmeticUnaryOpKind.Minus,
+ expr=e.build(),
+ ),
+ f"(-{e.reference})",
+ )
+
+
+def not_(e: Expr) -> Expr:
+ return Expr(
+ lambda: ast.UnaryOp(op=ast.UnaryOpKind.Not, expr=e.build()),
+ f"(NOT {e.reference})",
+ )
+
+
+def is_null(e: Expr, negated: bool) -> Expr:
+ return Expr(
+ lambda: ast.IsNull(expr=e.build(), negated=negated),
+ f"({e.reference} IS {'NOT ' if negated else ''}NULL)",
+ )
+
+
+def between(e: Expr, low: Expr, high: Expr, negated: bool) -> Expr:
+ return Expr(
+ lambda: ast.Between(
+ expr=e.build(),
+ low=low.build(),
+ high=high.build(),
+ negated=negated,
+ ),
+ f"({e.reference} {'NOT ' if negated else ''}BETWEEN "
+ f"{low.reference} AND {high.reference})",
+ )
+
+
+def function(name: str, *args: Expr) -> Expr:
+ return Expr(
+ lambda: ast.Function(ast.Name(name), args=[a.build() for a in args]),
+ f"{name}({', '.join(a.reference for a in args)})",
+ )
+
+
+def case(condition: Expr, result: Expr, otherwise: Expr) -> Expr:
+ return Expr(
+ lambda: ast.Case(
+ conditions=[condition.build()],
+ results=[result.build()],
+ else_result=otherwise.build(),
+ ),
+ f"(CASE WHEN {condition.reference} THEN {result.reference} "
+ f"ELSE {otherwise.reference} END)",
+ )
+
+
+ARITHMETIC = [ast.BinaryOpKind.Plus, ast.BinaryOpKind.Minus, ast.BinaryOpKind.Multiply]
+COMPARISON = [
+ ast.BinaryOpKind.Eq,
+ ast.BinaryOpKind.NotEq,
+ ast.BinaryOpKind.Lt,
+ ast.BinaryOpKind.GtEq,
+]
+LOGICAL = [ast.BinaryOpKind.And, ast.BinaryOpKind.Or]
+
+numeric_leaves = st.one_of(
+ st.sampled_from(["a", "b", "c"]).map(column),
+ st.integers(0, 3).map(number),
+ st.just(NULL),
+)
+
+
+base_comparison = st.builds(
+ binary,
+ st.sampled_from(COMPARISON),
+ numeric_leaves,
+ numeric_leaves,
+)
+numeric_atoms = st.one_of(
+ numeric_leaves,
+ st.builds(function, st.just("COALESCE"), numeric_leaves, numeric_leaves),
+ st.builds(case, base_comparison, numeric_leaves, numeric_leaves),
+)
+boolean_atoms = st.one_of(
+ st.builds(binary, st.sampled_from(COMPARISON), numeric_atoms, numeric_atoms),
+ st.builds(is_null, numeric_atoms, st.booleans()),
+ st.builds(between, numeric_atoms, numeric_leaves, numeric_leaves, st.booleans()),
+)
+expressions = st.one_of(
+ numeric_atoms,
+ st.builds(binary, st.sampled_from(ARITHMETIC), numeric_atoms, numeric_atoms),
+ numeric_atoms.map(negate),
+ boolean_atoms,
+ st.builds(binary, st.sampled_from(LOGICAL), boolean_atoms, boolean_atoms),
+ boolean_atoms.map(not_),
+)
+
+
+rows = st.lists(
+ st.tuples(*[st.one_of(st.none(), st.integers(-3, 3))] * 3),
+ min_size=1,
+ max_size=6,
+)
+
+
+def evaluate(conn, sql: str):
+ return conn.execute(f"SELECT {sql} FROM t ORDER BY rowid").fetchall()
+
+
+@pytest.fixture(scope="module")
+def conn():
+ with duckdb.connect(":memory:") as connection:
+ yield connection
+
+
+@settings(max_examples=examples(500))
+@given(expression=expressions, data=rows)
+def test_printed_tree_means_the_same_as_the_tree(conn, expression, data):
+ conn.execute("CREATE OR REPLACE TABLE t (a INTEGER, b INTEGER, c INTEGER)")
+ conn.executemany("INSERT INTO t VALUES (?, ?, ?)", data)
+ expected = evaluate(conn, expression.reference)
+
+ printed = str(expression.build())
+ try:
+ actual = evaluate(conn, printed)
+ except duckdb.Error as exc:
+ raise AssertionError(
+ f"DJ's SQL no longer runs.\n tree: {expression.reference}\n"
+ f" DJ: {printed}\n error: {exc}",
+ ) from exc
+ assert actual == expected, (
+ f"\n tree: {expression.reference}\n DJ: {printed}\n"
+ f" expected: {expected}\n got: {actual}"
+ )
+
+
+@PRINTER_DROPS_PARENTHESES.xfail()
+def test_known_issue_nested_negation_prints_as_comment(conn):
+ conn.execute("CREATE OR REPLACE TABLE t (a INTEGER, b INTEGER, c INTEGER)")
+ conn.execute("INSERT INTO t VALUES (1, 2, 3)")
+ expression = negate(negate(column("a")))
+ expected = evaluate(conn, expression.reference)
+ printed = str(expression.build())
+ try:
+ actual = evaluate(conn, printed)
+ except duckdb.Error as exc:
+ if printed.startswith("--"):
+ raise PropertyMismatch(
+ f"DJ printed {printed!r} for {expression!r}",
+ ) from exc
+ raise
+ assert actual == expected
diff --git a/datajunction-server/tests/property/budget.py b/datajunction-server/tests/property/budget.py
new file mode 100644
index 000000000..1a3810599
--- /dev/null
+++ b/datajunction-server/tests/property/budget.py
@@ -0,0 +1,19 @@
+"""
+How many examples each property runs.
+
+Properties declare a budget for an ordinary PR run; `DJ_PBT_PROFILE=nightly`
+multiplies it for longer searches.
+"""
+
+import os
+
+SCALE = {"dev": 1, "ci": 1, "nightly": 10}
+
+
+def profile() -> str:
+ default = "ci" if os.environ.get("CI") else "dev"
+ return os.environ.get("DJ_PBT_PROFILE", default)
+
+
+def examples(n: int) -> int:
+ return n * SCALE[profile()]
diff --git a/datajunction-server/tests/property/comparison.py b/datajunction-server/tests/property/comparison.py
new file mode 100644
index 000000000..5127b1800
--- /dev/null
+++ b/datajunction-server/tests/property/comparison.py
@@ -0,0 +1,18 @@
+"""Value comparison and failure type for property-test correctness oracles."""
+
+import math
+
+
+class PropertyMismatch(AssertionError):
+ """The result of a DJ operation differs from the independent answer."""
+
+
+def same_value(actual, expected) -> bool:
+ """Compare numbers approximately, but keep SQL NULL and non-finite values distinct."""
+ if actual is None or expected is None:
+ return actual is None and expected is None
+ if isinstance(actual, float) and math.isnan(actual):
+ return isinstance(expected, float) and math.isnan(expected)
+ if isinstance(expected, float) and math.isnan(expected):
+ return False
+ return math.isclose(actual, expected, rel_tol=1e-6, abs_tol=1e-6)
diff --git a/datajunction-server/tests/property/comparison_test.py b/datajunction-server/tests/property/comparison_test.py
new file mode 100644
index 000000000..6a98ca700
--- /dev/null
+++ b/datajunction-server/tests/property/comparison_test.py
@@ -0,0 +1,25 @@
+"""The oracle must not conflate distinct SQL result values."""
+
+import math
+
+import pytest
+
+from tests.property.comparison import same_value
+
+
+@pytest.mark.parametrize(
+ ("actual", "expected", "matches"),
+ [
+ (None, None, True),
+ (None, math.nan, False),
+ (None, math.inf, False),
+ (math.nan, math.nan, True),
+ (math.nan, math.inf, False),
+ (math.inf, math.inf, True),
+ (math.inf, -math.inf, False),
+ (1.0, 1.0 + 1e-7, True),
+ (1.0, 1.1, False),
+ ],
+)
+def test_same_value(actual, expected, matches):
+ assert same_value(actual, expected) is matches
diff --git a/datajunction-server/tests/property/conftest.py b/datajunction-server/tests/property/conftest.py
new file mode 100644
index 000000000..51d5285dd
--- /dev/null
+++ b/datajunction-server/tests/property/conftest.py
@@ -0,0 +1,34 @@
+"""
+Hypothesis profiles for the property tests, chosen with `DJ_PBT_PROFILE`:
+
+- dev (default locally): random examples; failures are saved and replayed.
+- ci (default when `CI` is set): a fixed sequence of examples and no shrinking,
+ so runs are repeatable and a failure is reported quickly.
+- nightly: ten times the examples, random, with shrinking.
+"""
+
+from hypothesis import HealthCheck, Phase, settings
+
+from tests.property.budget import profile
+
+COMMON = {
+ # Examples that create nodes through the API take well over the default
+ # 200ms, and vary with load.
+ "deadline": None,
+ "suppress_health_check": [
+ HealthCheck.function_scoped_fixture,
+ HealthCheck.too_slow,
+ ],
+ "print_blob": True,
+}
+
+settings.register_profile("dev", **COMMON)
+settings.register_profile(
+ "ci",
+ derandomize=True,
+ database=None,
+ phases=[Phase.explicit, Phase.reuse, Phase.generate],
+ **COMMON,
+)
+settings.register_profile("nightly", **COMMON)
+settings.load_profile(profile())
diff --git a/datajunction-server/tests/property/graph.py b/datajunction-server/tests/property/graph.py
new file mode 100644
index 000000000..f0a254823
--- /dev/null
+++ b/datajunction-server/tests/property/graph.py
@@ -0,0 +1,236 @@
+"""
+Random fact + dimension graphs, created in DJ through its API and mirrored as
+tables in DuckDB, for properties that compare DJ's SQL against other answers.
+"""
+
+import itertools
+
+from httpx import AsyncClient
+
+from tests.conftest import transpile_to_duckdb
+from tests.property.comparison import PropertyMismatch, same_value
+from tests.property.scenario import (
+ DimensionColumn,
+ FilterSpec,
+ Scenario,
+ render_filter,
+)
+
+counter = itertools.count()
+
+
+def accepts_missing_dimension(conn, filters: tuple[FilterSpec, ...]) -> bool:
+ """Whether all filters accept a fact with no matching dimension row."""
+ if not filters:
+ return True
+
+ def missing_column(column: DimensionColumn) -> str:
+ return "CAST(NULL AS VARCHAR)" if column == "label" else "CAST(NULL AS INTEGER)"
+
+ where = " AND ".join(
+ f"({render_filter(filter_, missing_column)})" for filter_ in filters
+ )
+ return bool(conn.sql(f"SELECT COALESCE(({where}), FALSE)").fetchone()[0])
+
+
+def has_orphan_facts(scenario: Scenario) -> bool:
+ keys = {key for key, _, _ in scenario.dimension_rows}
+ return any(key not in keys for key, _, _ in scenario.fact_rows)
+
+
+def by_group(rows, width: int) -> dict:
+ """Rows keyed by their first `width` columns (the requested dimensions)."""
+ grouped = {}
+ for row in rows:
+ key = row[:width]
+ assert key not in grouped, f"duplicate result group: {key}"
+ grouped[key] = row[width:]
+ return grouped
+
+
+def assert_same_results(actual: dict, expected: dict, context: str) -> None:
+ wrong = {
+ key: (actual.get(key, ""), expected.get(key, ""))
+ for key in actual.keys() | expected.keys()
+ if key not in actual
+ or key not in expected
+ or len(actual[key]) != len(expected[key])
+ or not all(same_value(a, e) for a, e in zip(actual[key], expected[key]))
+ }
+ if wrong:
+ raise PropertyMismatch(f"\n{context}\n(actual, expected) by group: {wrong}")
+
+
+class Graph:
+ def __init__(self, n: int):
+ self.fact_table, self.dim_table = f"fact_{n}", f"dim_{n}"
+ self.fact = f"default.pbt{n}_fact"
+ self.dim_src = f"default.pbt{n}_dim_src"
+ self.dim = f"default.pbt{n}_dim"
+ self.prefix = f"default.pbt{n}_"
+
+ def dimension(self, column: str) -> str:
+ return f"{self.dim}.{column}"
+
+
+async def post(client: AsyncClient, path: str, payload: dict):
+ response = await client.post(path, json=payload)
+ assert response.status_code in (200, 201), (path, response.json())
+ return response.json()
+
+
+async def build_graph(client, conn, scenario: Scenario) -> Graph:
+ graph = Graph(next(counter))
+ conn.execute('CREATE SCHEMA IF NOT EXISTS "default".pbt')
+ conn.execute(
+ f'CREATE TABLE "default".pbt.{graph.dim_table} '
+ "(k INTEGER, label VARCHAR, bucket INTEGER)",
+ )
+ conn.execute(
+ f'CREATE TABLE "default".pbt.{graph.fact_table} '
+ "(k INTEGER, x INTEGER, y INTEGER)",
+ )
+ conn.executemany(
+ f'INSERT INTO "default".pbt.{graph.dim_table} VALUES (?, ?, ?)',
+ scenario.dimension_rows,
+ )
+ if scenario.fact_rows:
+ conn.executemany(
+ f'INSERT INTO "default".pbt.{graph.fact_table} VALUES (?, ?, ?)',
+ scenario.fact_rows,
+ )
+
+ def source(name, table, columns):
+ return {
+ "name": name,
+ "catalog": "default",
+ "schema_": "pbt",
+ "table": table,
+ "mode": "published",
+ "columns": [{"name": c, "type": t} for c, t in columns],
+ }
+
+ await post(
+ client,
+ "/nodes/source/",
+ source(
+ graph.fact,
+ graph.fact_table,
+ [("k", "int"), ("x", "int"), ("y", "int")],
+ ),
+ )
+ await post(
+ client,
+ "/nodes/source/",
+ source(
+ graph.dim_src,
+ graph.dim_table,
+ [("k", "int"), ("label", "string"), ("bucket", "int")],
+ ),
+ )
+ await post(
+ client,
+ "/nodes/dimension/",
+ {
+ "name": graph.dim,
+ "query": f"SELECT k, label, bucket FROM {graph.dim_src}",
+ "primary_key": ["k"],
+ "mode": "published",
+ },
+ )
+ await post(
+ client,
+ f"/nodes/{graph.fact}/link",
+ {
+ "dimension_node": graph.dim,
+ "join_type": scenario.join_type,
+ "join_on": f"{graph.fact}.k = {graph.dim}.k",
+ },
+ )
+ return graph
+
+
+async def create_metric(client, graph: Graph, name: str, query: str) -> str:
+ metric = graph.prefix + name
+ await post(
+ client,
+ "/nodes/metric/",
+ {"name": metric, "query": query, "mode": "published"},
+ )
+ return metric
+
+
+async def metrics_sql(
+ client,
+ conn,
+ graph: Graph,
+ scenario: Scenario,
+ metrics: list[str],
+ *,
+ use_materialized: bool = True,
+ parenthesize_filters: bool = True,
+) -> tuple[str, dict]:
+ """DJ's SQL for the scenario's request, and its result keyed by dimensions."""
+ filters = [
+ render_filter(filter_, lambda column: graph.dimension(column))
+ for filter_ in scenario.filters
+ ]
+ if parenthesize_filters:
+ # Combining an unparenthesized OR with another filter is
+ # known_issues.PRINTER_DROPS_PARENTHESES.
+ filters = [f"({f})" for f in filters]
+ response = await client.get(
+ "/sql/metrics/v3/",
+ params={
+ "metrics": metrics,
+ "dimensions": [graph.dimension(d) for d in scenario.dimensions],
+ "filters": filters,
+ "use_materialized": use_materialized,
+ },
+ )
+ assert response.status_code == 200, response.json()
+ sql = response.json()["sql"]
+ rows = conn.sql(transpile_to_duckdb(sql)).fetchall()
+ return sql, by_group(rows, len(scenario.dimensions))
+
+
+def reference(
+ conn,
+ graph: Graph,
+ scenario: Scenario,
+ value_expression: str,
+) -> tuple[str, dict]:
+ """A plain join-filter-group query over the raw tables."""
+ group_cols = [f"d.{column}" for column in scenario.dimensions]
+ join = (
+ f' {scenario.join_type.upper()} JOIN "default".pbt.{graph.dim_table} d '
+ "ON f.k = d.k"
+ if scenario.dimensions or scenario.filters
+ else ""
+ )
+ where = (
+ " WHERE "
+ + " AND ".join(
+ f"({render_filter(filter_, lambda column: f'd.{column}')})"
+ for filter_ in scenario.filters
+ )
+ if scenario.filters
+ else ""
+ )
+ group_by = f" GROUP BY {', '.join(group_cols)}" if group_cols else ""
+ sql = (
+ f"SELECT {', '.join([*group_cols, value_expression])} "
+ f'FROM "default".pbt.{graph.fact_table} f{join}{where}{group_by}'
+ )
+ return sql, by_group(conn.sql(sql).fetchall(), len(group_cols))
+
+
+def describe(scenario: Scenario, *sql: str) -> str:
+ return "\n".join(
+ [
+ f"metrics: {[m.definition for m in scenario.metrics]} join: {scenario.join_type}",
+ f"dimensions: {scenario.dimensions} "
+ f"filters: {[render_filter(f, str) for f in scenario.filters]}",
+ *sql,
+ ],
+ )
diff --git a/datajunction-server/tests/property/known_issues.py b/datajunction-server/tests/property/known_issues.py
new file mode 100644
index 000000000..d7327b514
--- /dev/null
+++ b/datajunction-server/tests/property/known_issues.py
@@ -0,0 +1,64 @@
+"""
+Bugs the property tests have found and that are not yet fixed.
+
+Generated tests exclude cases affected by known bugs, so they can keep
+searching. Each bug also has a small repro marked `xfail(strict=True)`: when
+the bug is fixed the repro passes, pytest reports it as XPASS and fails, and
+that is the cue to delete the entry here along with its exclusions.
+"""
+
+from dataclasses import dataclass
+
+import pytest
+
+from tests.property.comparison import PropertyMismatch
+
+
+@dataclass(frozen=True)
+class KnownIssue:
+ summary: str
+ issue: str | None = None # link once filed
+
+ @property
+ def reason(self) -> str:
+ return self.summary + (f" ({self.issue})" if self.issue else "")
+
+ def xfail(self):
+ # Only the expected oracle mismatch is an XFAIL. Setup/API assertions
+ # and unrelated exceptions must still fail the test.
+ return pytest.mark.xfail(
+ reason=self.reason,
+ strict=True,
+ raises=PropertyMismatch,
+ )
+
+ def skip(self):
+ return pytest.mark.skip(reason=f"known issue: {self.reason}")
+
+
+PRINTER_DROPS_PARENTHESES = KnownIssue(
+ "SQL assembled from AST nodes built in code prints without the parentheses "
+ "its nesting needs: sample variance/stddev/correlation combiners, combined "
+ "filters, derived metrics over compound metrics, and `-(-x)` as `--x`",
+)
+COVARIANCE_COUNTS_HALF_NULL_ROWS = KnownIssue(
+ "COVAR_POP/COVAR_SAMP/CORR decompositions count rows where only one of the "
+ "two arguments is NULL",
+)
+COUNT_COMPONENT_NAME_COLLISION = KnownIssue(
+ "COUNT(x) from AVG/COUNT/VAR and from COVAR/CORR share a component name but "
+ "record the aggregation as `COUNT` vs `COUNT(x)`, so frozen measures conflict",
+)
+EMPTY_PREAGG_COUNT_IS_NULL = KnownIssue(
+ "a COUNT served from a pre-aggregation with no matching rows is NULL rather than 0",
+)
+FILTER_PUSHED_PAST_LEFT_JOIN = KnownIssue(
+ "a dimension filter is also applied inside the LEFT-joined dimension, so a "
+ "filter that accepts NULL (e.g. `label IS NULL`) matches facts whose "
+ "dimension row it filtered out",
+)
+PREAGG_INNER_JOIN_DROPS_ORPHANS = KnownIssue(
+ "a pre-aggregation planned with an INNER-linked dimension drops facts with "
+ "no matching dimension row, so requests that don't use that dimension get "
+ "a different answer when served from it",
+)
diff --git a/datajunction-server/tests/property/materialization_test.py b/datajunction-server/tests/property/materialization_test.py
new file mode 100644
index 000000000..2b7ac9ced
--- /dev/null
+++ b/datajunction-server/tests/property/materialization_test.py
@@ -0,0 +1,228 @@
+"""
+Property: serving a request from a pre-aggregation gives the same answer as
+computing it from the raw tables.
+
+Each example plans a pre-aggregation at a grain covering the request's
+dimensions and filters, materializes DJ's planned SQL on DuckDB, reports it
+available, then asks for SQL with and without `use_materialized`.
+
+Both answers come from DJ, so a bug that affects both paths the same way is
+invisible here; `random_graph_test` compares against a reference query instead.
+"""
+
+import duckdb
+import pytest
+from httpx import AsyncClient
+from hypothesis import assume, event, given, settings
+
+from tests.conftest import transpile_to_duckdb
+from tests.property.budget import examples
+from tests.property.graph import (
+ assert_same_results,
+ build_graph,
+ create_metric,
+ describe,
+ has_orphan_facts,
+ metrics_sql,
+ post,
+)
+from tests.property.known_issues import (
+ EMPTY_PREAGG_COUNT_IS_NULL,
+ PREAGG_INNER_JOIN_DROPS_ORPHANS,
+)
+from tests.property.scenario import (
+ METRIC_BY_DEFINITION,
+ Scenario,
+ materialization_scenarios,
+)
+
+
+async def check_preaggregated(
+ client,
+ conn,
+ scenario: Scenario,
+ *,
+ skip_known: bool,
+ discard_if_not_served: bool = False,
+):
+ preagg_grain = scenario.preagg_grain
+ if preagg_grain is None:
+ raise ValueError("a materialization scenario needs a pre-aggregation grain")
+ graph = await build_graph(client, conn, scenario)
+ metrics = [
+ await create_metric(
+ client,
+ graph,
+ f"m{i}",
+ f"SELECT {query.definition} FROM {graph.fact}",
+ )
+ for i, query in enumerate(scenario.metrics)
+ ]
+
+ plan = await post(
+ client,
+ "/preaggs/plan/",
+ {
+ "metrics": metrics,
+ "dimensions": [graph.dimension(d) for d in preagg_grain],
+ },
+ )
+ tables = []
+ for i, preagg in enumerate(plan["preaggs"]):
+ table = f"{graph.fact_table}_preagg_{i}"
+ conn.execute(
+ f'CREATE TABLE "default".pbt.{table} AS '
+ f"{transpile_to_duckdb(preagg['sql'])}",
+ )
+ await post(
+ client,
+ f"/preaggs/{preagg['id']}/availability/",
+ {
+ "catalog": "default",
+ "schema_": "pbt",
+ "table": table,
+ "valid_through_ts": 20990101,
+ },
+ )
+ tables.append(table)
+
+ preagg_sql, preaggregated = await metrics_sql(
+ client,
+ conn,
+ graph,
+ scenario,
+ metrics,
+ )
+ raw_sql, raw = await metrics_sql(
+ client,
+ conn,
+ graph,
+ scenario,
+ metrics,
+ use_materialized=False,
+ )
+ served_from_preagg = any(table in preagg_sql for table in tables)
+ if discard_if_not_served:
+ # Some otherwise compatible requests still cannot use the planned
+ # pre-aggregation; retain only examples that exercise substitution.
+ event(f"served from pre-aggregation: {served_from_preagg}")
+ assume(served_from_preagg)
+ else:
+ assert served_from_preagg, "the planned pre-aggregation was not used"
+
+ if skip_known and () in preaggregated and () in raw:
+ assume(
+ not any(p is None and r == 0 for p, r in zip(preaggregated[()], raw[()])),
+ )
+ assert_same_results(
+ preaggregated,
+ raw,
+ describe(
+ scenario,
+ f"pre-aggregation grain: {preagg_grain} used: {served_from_preagg}",
+ "planned SQL:\n" + "\n---\n".join(p["sql"] for p in plan["preaggs"]),
+ f"SQL using the pre-aggregation:\n{preagg_sql}",
+ f"SQL from raw tables:\n{raw_sql}",
+ ),
+ )
+
+
+@pytest.mark.asyncio
+@settings(max_examples=examples(15))
+@given(scenario=materialization_scenarios())
+async def test_preaggregated_matches_raw(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+ scenario: Scenario,
+):
+ assume(
+ not (
+ scenario.join_type == "inner"
+ and scenario.preagg_grain
+ and not scenario.dimensions
+ and not scenario.filters
+ and has_orphan_facts(scenario)
+ ),
+ ) # known_issues.PREAGG_INNER_JOIN_DROPS_ORPHANS
+ await check_preaggregated(
+ module__client_with_roads,
+ duckdb_conn,
+ scenario,
+ skip_known=True,
+ discard_if_not_served=True,
+ )
+
+
+@pytest.mark.asyncio
+async def test_eligible_preaggregation_is_used_and_matches_raw(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+):
+ """A compatible rollup must route through the available pre-aggregation."""
+ await check_preaggregated(
+ module__client_with_roads,
+ duckdb_conn,
+ Scenario(
+ dimension_rows=((0, "a", 1), (1, "b", 2)),
+ fact_rows=((0, 1, 2), (0, 2, 3), (1, 3, 4)),
+ metrics=(
+ METRIC_BY_DEFINITION["SUM(x)"],
+ METRIC_BY_DEFINITION["COUNT(x)"],
+ ),
+ op="+",
+ join_type="left",
+ dimensions=(),
+ filters=(),
+ preagg_grain=("label",),
+ ),
+ skip_known=False,
+ )
+
+
+@pytest.mark.asyncio
+@EMPTY_PREAGG_COUNT_IS_NULL.xfail()
+async def test_known_issue_count_from_empty_preaggregation(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+):
+ await check_preaggregated(
+ module__client_with_roads,
+ duckdb_conn,
+ Scenario(
+ dimension_rows=((0, None, None),),
+ fact_rows=(),
+ metrics=(
+ METRIC_BY_DEFINITION["SUM(x)"],
+ METRIC_BY_DEFINITION["COUNT(x)"],
+ ),
+ op="+",
+ join_type="left",
+ dimensions=(),
+ filters=(),
+ preagg_grain=("label",),
+ ),
+ skip_known=False,
+ )
+
+
+@pytest.mark.asyncio
+@PREAGG_INNER_JOIN_DROPS_ORPHANS.xfail()
+async def test_known_issue_inner_linked_preaggregation_drops_orphan_facts(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+):
+ await check_preaggregated(
+ module__client_with_roads,
+ duckdb_conn,
+ Scenario(
+ dimension_rows=((0, "a", 1),),
+ fact_rows=((0, 1, 0), (None, 5, 0)),
+ metrics=(METRIC_BY_DEFINITION["SUM(x)"],) * 2,
+ op="+",
+ join_type="inner",
+ dimensions=(),
+ filters=(),
+ preagg_grain=("label",),
+ ),
+ skip_known=False,
+ )
diff --git a/datajunction-server/tests/property/metric_decomposition_test.py b/datajunction-server/tests/property/metric_decomposition_test.py
new file mode 100644
index 000000000..e7032b90a
--- /dev/null
+++ b/datajunction-server/tests/property/metric_decomposition_test.py
@@ -0,0 +1,180 @@
+"""
+Property: a decomposed metric, rolled up from a finer grain, equals the metric
+computed directly.
+
+DJ's own decomposition (`MetricComponentExtractor._extract_base`) and component
+rendering (`build_component_expression`) produce the SQL; DuckDB executes both
+that SQL and the plain aggregate, and the two answers must agree.
+"""
+
+import duckdb
+import pytest
+import sqlglot
+from hypothesis import assume, given, settings
+from hypothesis import strategies as st
+
+from datajunction_server.construction.build_v3.decomposition import (
+ build_component_expression,
+)
+from datajunction_server.sql.decompose import MetricComponentExtractor
+from datajunction_server.sql.parsing.backends.antlr4 import parse
+from tests.property.budget import examples
+from tests.property.comparison import PropertyMismatch, same_value
+from tests.property.known_issues import (
+ COUNT_COMPONENT_NAME_COLLISION,
+ COVARIANCE_COUNTS_HALF_NULL_ROWS,
+ PRINTER_DROPS_PARENTHESES,
+)
+
+METRICS = [
+ "SUM(x)",
+ "MIN(x)",
+ "MAX(x)",
+ "COUNT(x)",
+ "COUNT(*)",
+ "COUNT_IF(x > 0)",
+ "AVG(x)",
+ "VAR_POP(x)",
+ "VAR_SAMP(x)",
+ "VARIANCE(x)",
+ "STDDEV_POP(x)",
+ "STDDEV_SAMP(x)",
+ "STDDEV(x)",
+ "COVAR_POP(x, y)",
+ "COVAR_SAMP(x, y)",
+ "CORR(x, y)",
+]
+BROKEN = {
+ "VAR_SAMP(x)": PRINTER_DROPS_PARENTHESES,
+ "VARIANCE(x)": PRINTER_DROPS_PARENTHESES,
+ "STDDEV_SAMP(x)": PRINTER_DROPS_PARENTHESES,
+ "STDDEV(x)": PRINTER_DROPS_PARENTHESES,
+ "COVAR_POP(x, y)": COVARIANCE_COUNTS_HALF_NULL_ROWS,
+ "COVAR_SAMP(x, y)": COVARIANCE_COUNTS_HALF_NULL_ROWS,
+ "CORR(x, y)": COVARIANCE_COUNTS_HALF_NULL_ROWS,
+}
+
+values = st.one_of(st.none(), st.integers(-100, 100))
+rows_strategy = st.lists(
+ st.tuples(st.integers(0, 2), st.integers(0, 3), values, values),
+ min_size=1,
+ max_size=40,
+)
+
+
+def to_duckdb(spark_sql: str) -> str:
+ return sqlglot.transpile(spark_sql, read="spark", write="duckdb")[0]
+
+
+def decomposed_sql(metric: str) -> tuple[str, str]:
+ """DJ's measures query (grain g, p) and its combiner rolled up to grain g."""
+ components, derived = MetricComponentExtractor(0)._extract_base(
+ parse(f"SELECT {metric} FROM t"),
+ )
+ measures = ", ".join(
+ f"{build_component_expression(c)} AS {c.name}" for c in components
+ )
+ measures_sql = f"SELECT g, p, {measures} FROM t GROUP BY g, p"
+ combiner = str(derived.select.projection[0])
+ return measures_sql, f"SELECT g, {combiner} AS v FROM m GROUP BY g"
+
+
+@pytest.fixture(scope="module")
+def conn():
+ with duckdb.connect(":memory:") as connection:
+ yield connection
+
+
+def check_rollup(conn, metric, column_type, rows):
+ conn.execute(
+ f"CREATE OR REPLACE TABLE t (g INTEGER, p INTEGER, "
+ f"x {column_type}, y {column_type})",
+ )
+ conn.executemany("INSERT INTO t VALUES (?, ?, ?, ?)", rows)
+
+ measures_sql, combine_sql = decomposed_sql(metric)
+ conn.execute(f"CREATE OR REPLACE TABLE m AS {to_duckdb(measures_sql)}")
+ rolled_up = dict(conn.execute(to_duckdb(combine_sql)).fetchall())
+ direct = dict(
+ conn.execute(f"SELECT g, {metric} FROM t GROUP BY g").fetchall(),
+ )
+
+ assert rolled_up.keys() == direct.keys()
+ mismatches = {
+ g: (rolled_up[g], direct[g])
+ for g in direct
+ if not same_value(rolled_up[g], direct[g])
+ }
+ if mismatches:
+ raise PropertyMismatch(
+ f"{metric}: (rolled-up, direct) by group = {mismatches}\n"
+ f"measures: {measures_sql}\ncombine: {combine_sql}",
+ )
+
+
+@pytest.mark.parametrize("column_type", ["INTEGER", "DOUBLE"])
+@pytest.mark.parametrize(
+ "metric",
+ [pytest.param(m, marks=BROKEN[m].skip()) if m in BROKEN else m for m in METRICS],
+)
+@settings(max_examples=examples(100))
+@given(rows=rows_strategy)
+def test_rollup_matches_direct_aggregate(conn, metric, column_type, rows):
+ check_rollup(conn, metric, column_type, rows)
+
+
+@PRINTER_DROPS_PARENTHESES.xfail()
+def test_known_issue_sample_variance_rollup(conn):
+ check_rollup(conn, "VAR_SAMP(x)", "INTEGER", [(0, 0, 0, None), (0, 0, 1, None)])
+
+
+@COVARIANCE_COUNTS_HALF_NULL_ROWS.xfail()
+def test_known_issue_covariance_with_one_side_null(conn):
+ check_rollup(conn, "COVAR_POP(x, y)", "INTEGER", [(0, 0, 1, None), (0, 0, 0, 1)])
+
+
+def components_of(metric: str):
+ components, _ = MetricComponentExtractor(0)._extract_base(
+ parse(f"SELECT {metric} FROM t"),
+ )
+ return components
+
+
+def check_component_names(first: str, second: str, *, skip_known: bool) -> None:
+ by_name = {c.name: c for c in components_of(first)}
+ for component in components_of(second):
+ other = by_name.get(component.name)
+ if other is None:
+ continue
+ if skip_known:
+ spelled_differently = {other.aggregation, component.aggregation} == {
+ "COUNT",
+ f"COUNT({component.expression})",
+ }
+ assume(not spelled_differently)
+ if (other.expression, other.aggregation, other.merge) != (
+ component.expression,
+ component.aggregation,
+ component.merge,
+ ):
+ raise PropertyMismatch(
+ f"`{component.name}`: from {first} it is aggregation="
+ f"{other.aggregation!r} over {other.expression!r}; from {second} it is "
+ f"aggregation={component.aggregation!r} over {component.expression!r}",
+ )
+
+
+@settings(max_examples=examples(200))
+@given(first=st.sampled_from(METRICS), second=st.sampled_from(METRICS))
+def test_same_component_name_means_same_component(first, second):
+ """
+ Frozen measures are keyed by component name, so two metrics on the same
+ parent that produce a component with the same name must define it the same
+ way, or registering the second metric's measures fails.
+ """
+ check_component_names(first, second, skip_known=True)
+
+
+@COUNT_COMPONENT_NAME_COLLISION.xfail()
+def test_known_issue_count_component_collision():
+ check_component_names("COUNT(x)", "COVAR_POP(x, y)", skip_known=False)
diff --git a/datajunction-server/tests/property/random_graph_test.py b/datajunction-server/tests/property/random_graph_test.py
new file mode 100644
index 000000000..d1bbff5d2
--- /dev/null
+++ b/datajunction-server/tests/property/random_graph_test.py
@@ -0,0 +1,221 @@
+"""
+Property: for a randomly generated fact + dimension graph, the SQL DJ builds for
+a metric request returns the same rows as a plain hand-written query over the
+raw tables.
+"""
+
+from dataclasses import replace
+
+import duckdb
+import pytest
+from httpx import AsyncClient
+from hypothesis import assume, given, settings
+
+from tests.property.budget import examples
+from tests.property.graph import (
+ accepts_missing_dimension,
+ assert_same_results,
+ build_graph,
+ create_metric,
+ describe,
+ metrics_sql,
+ reference,
+)
+from tests.property.known_issues import (
+ FILTER_PUSHED_PAST_LEFT_JOIN,
+ PRINTER_DROPS_PARENTHESES,
+)
+from tests.property.scenario import (
+ METRIC_BY_DEFINITION,
+ SIMPLE_METRICS,
+ FilterAtom,
+ OrFilter,
+ Scenario,
+ scenarios,
+)
+
+
+def skip_known_filter_pushdown(conn, scenario: Scenario):
+ """known_issues.FILTER_PUSHED_PAST_LEFT_JOIN"""
+ assume(
+ not (
+ scenario.join_type == "left"
+ and scenario.filters
+ and accepts_missing_dimension(conn, scenario.filters)
+ ),
+ )
+
+
+async def check_base_metric(client, conn, scenario, *, parenthesize_filters=True):
+ graph = await build_graph(client, conn, scenario)
+ query = scenario.metrics[0]
+ metric = await create_metric(
+ client,
+ graph,
+ "metric",
+ f"SELECT {query.definition} FROM {graph.fact}",
+ )
+
+ dj_sql, dj = await metrics_sql(
+ client,
+ conn,
+ graph,
+ scenario,
+ [metric],
+ parenthesize_filters=parenthesize_filters,
+ )
+ reference_sql, expected = reference(conn, graph, scenario, query.reference)
+ assert_same_results(dj, expected, describe(scenario, reference_sql, dj_sql))
+
+
+async def check_derived_metric(client, conn, scenario):
+ graph = await build_graph(client, conn, scenario)
+ first, second = scenario.metrics
+ op = scenario.op
+ m1 = await create_metric(
+ client,
+ graph,
+ "m1",
+ f"SELECT {first.definition} FROM {graph.fact}",
+ )
+ m2 = await create_metric(
+ client,
+ graph,
+ "m2",
+ f"SELECT {second.definition} FROM {graph.fact}",
+ )
+ derived = await create_metric(client, graph, "derived", f"SELECT {m1} {op} {m2}")
+
+ dj_sql, dj = await metrics_sql(client, conn, graph, scenario, [derived])
+ # DJ defines metric division as NULL on a zero denominator. Keep that
+ # contract in the separately written reference query, and compare NULL
+ # distinctly from infinity or NaN in assert_same_results.
+ right = f"NULLIF({second.reference}, 0)" if op == "/" else second.reference
+ reference_sql, expected = reference(
+ conn,
+ graph,
+ scenario,
+ f"({first.reference}) {op} ({right})",
+ )
+ assert_same_results(
+ dj,
+ expected,
+ describe(scenario, f"derived: m1 {op} m2", reference_sql, dj_sql),
+ )
+ return dj
+
+
+@pytest.mark.asyncio
+@settings(max_examples=examples(20))
+@given(scenario=scenarios())
+async def test_base_metric_matches_reference_query(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+ scenario,
+):
+ skip_known_filter_pushdown(duckdb_conn, scenario)
+ await check_base_metric(module__client_with_roads, duckdb_conn, scenario)
+
+
+@pytest.mark.asyncio
+@settings(max_examples=examples(20))
+@given(scenario=scenarios(metric_pool=SIMPLE_METRICS))
+async def test_derived_metric_matches_reference_query(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+ scenario,
+):
+ skip_known_filter_pushdown(duckdb_conn, scenario)
+ await check_derived_metric(module__client_with_roads, duckdb_conn, scenario)
+
+
+def fixed(**overrides) -> Scenario:
+ scenario = Scenario(
+ dimension_rows=((0, "a", 0), (1, "b", 1)),
+ fact_rows=((0, 1, 0), (1, 2, 0)),
+ metrics=(METRIC_BY_DEFINITION["SUM(x)"],) * 2,
+ op="+",
+ join_type="left",
+ dimensions=("label",),
+ filters=(),
+ )
+ return replace(scenario, **overrides)
+
+
+@pytest.mark.asyncio
+async def test_derived_metric_division_by_zero_returns_null(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+):
+ result = await check_derived_metric(
+ module__client_with_roads,
+ duckdb_conn,
+ fixed(
+ fact_rows=((0, 1, 0),),
+ metrics=(
+ METRIC_BY_DEFINITION["COUNT(*)"],
+ METRIC_BY_DEFINITION["STDDEV_POP(x)"],
+ ),
+ op="/",
+ dimensions=(),
+ ),
+ )
+ assert result[()] == (None,)
+
+
+@pytest.mark.asyncio
+@PRINTER_DROPS_PARENTHESES.xfail()
+async def test_known_issue_or_filter_combined_with_another_filter(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+):
+ await check_base_metric(
+ module__client_with_roads,
+ duckdb_conn,
+ fixed(
+ filters=(
+ OrFilter(FilterAtom("label", "=", "a"), FilterAtom("label", "=", "b")),
+ FilterAtom("label", "=", "b"),
+ ),
+ ),
+ parenthesize_filters=False,
+ )
+
+
+@pytest.mark.asyncio
+@PRINTER_DROPS_PARENTHESES.xfail()
+async def test_known_issue_derived_metric_over_compound_metric(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+):
+ await check_derived_metric(
+ module__client_with_roads,
+ duckdb_conn,
+ fixed(
+ fact_rows=((None, -4, 0),),
+ metrics=(
+ METRIC_BY_DEFINITION["MAX(x) - MIN(y)"],
+ METRIC_BY_DEFINITION["SUM(x)"],
+ ),
+ op="/",
+ dimensions=(),
+ ),
+ )
+
+
+@pytest.mark.asyncio
+@FILTER_PUSHED_PAST_LEFT_JOIN.xfail()
+async def test_known_issue_null_accepting_filter_on_left_joined_dimension(
+ module__client_with_roads: AsyncClient,
+ duckdb_conn: duckdb.DuckDBPyConnection,
+):
+ await check_base_metric(
+ module__client_with_roads,
+ duckdb_conn,
+ fixed(
+ dimension_rows=((0, "b", None), (1, None, None)),
+ fact_rows=((0, 3, None), (1, 5, None)),
+ dimensions=("label",),
+ filters=(FilterAtom("label", "IS NULL"),),
+ ),
+ )
diff --git a/datajunction-server/tests/property/scenario.py b/datajunction-server/tests/property/scenario.py
new file mode 100644
index 000000000..0f72c18e3
--- /dev/null
+++ b/datajunction-server/tests/property/scenario.py
@@ -0,0 +1,203 @@
+"""Small, immutable cases shared by the graph and materialization properties."""
+
+from __future__ import annotations
+
+from dataclasses import dataclass, replace
+from typing import Literal, TypeAlias
+from collections.abc import Callable
+
+from hypothesis import strategies as st
+
+DimensionColumn: TypeAlias = Literal["label", "bucket"]
+DimensionRow: TypeAlias = tuple[int, str | None, int | None]
+FactRow: TypeAlias = tuple[int | None, int | None, int | None]
+FilterValue: TypeAlias = str | int
+
+
+@dataclass(frozen=True)
+class MetricSpec:
+ """DJ definition and separately written raw-table reference expression."""
+
+ definition: str
+ reference: str
+ compound_combiner: bool = False
+
+
+# The reference expressions are explicit. Deriving them by editing DJ SQL would
+# let a qualification error in the test's oracle hide a DJ error.
+# Sample variance/stddev and covariance/correlation remain in fixed known-bug
+# repros until their decomposition and printing issues are resolved.
+METRICS = (
+ MetricSpec("SUM(x)", "SUM(f.x)"),
+ MetricSpec("COUNT(x)", "COUNT(f.x)"),
+ MetricSpec("COUNT(*)", "COUNT(*)"),
+ MetricSpec("MIN(x)", "MIN(f.x)"),
+ MetricSpec("MAX(x)", "MAX(f.x)"),
+ MetricSpec("AVG(x)", "AVG(f.x)", compound_combiner=True),
+ MetricSpec("VAR_POP(x)", "VAR_POP(f.x)", compound_combiner=True),
+ MetricSpec("STDDEV_POP(x)", "STDDEV_POP(f.x)"),
+ MetricSpec("COUNT(DISTINCT x)", "COUNT(DISTINCT f.x)"),
+ MetricSpec("SUM(x + y)", "SUM(f.x + f.y)"),
+ MetricSpec(
+ "SUM(CASE WHEN x > 0 THEN x END)",
+ "SUM(CASE WHEN f.x > 0 THEN f.x END)",
+ ),
+ MetricSpec("AVG(x) * 2", "AVG(f.x) * 2", compound_combiner=True),
+ MetricSpec(
+ "SUM(x) / COUNT(*)",
+ "SUM(f.x) / COUNT(*)",
+ compound_combiner=True,
+ ),
+ MetricSpec(
+ "MAX(x) - MIN(y)",
+ "MAX(f.x) - MIN(f.y)",
+ compound_combiner=True,
+ ),
+)
+METRIC_BY_DEFINITION = {metric.definition: metric for metric in METRICS}
+# Derived metrics currently lose parentheses around compound combiners.
+SIMPLE_METRICS = tuple(metric for metric in METRICS if not metric.compound_combiner)
+
+
+@dataclass(frozen=True)
+class FilterAtom:
+ column: DimensionColumn
+ operator: Literal["=", "<>", ">", "IS NULL", "IS NOT NULL", "IN"]
+ value: FilterValue | tuple[FilterValue, ...] | None = None
+
+
+@dataclass(frozen=True)
+class OrFilter:
+ left: FilterSpec
+ right: FilterSpec
+
+
+@dataclass(frozen=True)
+class NotFilter:
+ inner: FilterSpec
+
+
+FilterSpec: TypeAlias = FilterAtom | OrFilter | NotFilter
+
+
+def _literal(value: FilterValue) -> str:
+ if isinstance(value, str):
+ return "'" + value.replace("'", "''") + "'"
+ return str(value)
+
+
+def render_filter(
+ filter_: FilterSpec,
+ column_ref: Callable[[DimensionColumn], str],
+) -> str:
+ """Render a predicate with DJ names or physical table aliases."""
+ if isinstance(filter_, OrFilter):
+ return (
+ f"{render_filter(filter_.left, column_ref)} OR "
+ f"{render_filter(filter_.right, column_ref)}"
+ )
+ if isinstance(filter_, NotFilter):
+ return f"NOT ({render_filter(filter_.inner, column_ref)})"
+
+ column = column_ref(filter_.column)
+ if filter_.operator in ("IS NULL", "IS NOT NULL"):
+ return f"{column} {filter_.operator}"
+ if filter_.operator == "IN":
+ assert isinstance(filter_.value, tuple)
+ values = ", ".join(_literal(value) for value in filter_.value)
+ return f"{column} IN ({values})"
+ assert isinstance(filter_.value, (str, int))
+ return f"{column} {filter_.operator} {_literal(filter_.value)}"
+
+
+def filter_columns(filter_: FilterSpec) -> set[DimensionColumn]:
+ """Dimension columns a predicate needs to remain available after rollup."""
+ if isinstance(filter_, OrFilter):
+ return filter_columns(filter_.left) | filter_columns(filter_.right)
+ if isinstance(filter_, NotFilter):
+ return filter_columns(filter_.inner)
+ return {filter_.column}
+
+
+@dataclass(frozen=True)
+class Scenario:
+ dimension_rows: tuple[DimensionRow, ...]
+ fact_rows: tuple[FactRow, ...]
+ metrics: tuple[MetricSpec, MetricSpec]
+ op: Literal["+", "-", "*", "/"]
+ join_type: Literal["left", "inner"]
+ dimensions: tuple[DimensionColumn, ...]
+ filters: tuple[FilterSpec, ...]
+ preagg_grain: tuple[DimensionColumn, ...] | None = None
+
+
+DIMENSION_COLUMNS: tuple[DimensionColumn, ...] = ("label", "bucket")
+ATOMS = (
+ FilterAtom("label", "=", "a"),
+ FilterAtom("label", "<>", "b"),
+ FilterAtom("label", "IS NULL"),
+ FilterAtom("label", "IS NOT NULL"),
+ FilterAtom("label", "IN", ("a", "c")),
+ FilterAtom("bucket", "=", 1),
+ FilterAtom("bucket", ">", 0),
+ FilterAtom("bucket", "IS NULL"),
+ FilterAtom("bucket", "IN", (0, 2)),
+)
+
+labels = st.one_of(st.none(), st.sampled_from(["a", "b", "c"]))
+buckets = st.one_of(st.none(), st.integers(0, 2))
+values = st.one_of(st.none(), st.integers(-5, 5))
+atom = st.sampled_from(ATOMS)
+filters = st.one_of(
+ atom,
+ st.tuples(atom, atom).map(lambda pair: OrFilter(*pair)),
+ atom.map(NotFilter),
+)
+dimension_subsets = st.lists(
+ st.sampled_from(DIMENSION_COLUMNS),
+ unique=True,
+ max_size=len(DIMENSION_COLUMNS),
+).map(tuple)
+
+
+@st.composite
+def scenarios(draw, metric_pool=METRICS):
+ n_dims = draw(st.integers(1, 3))
+ return Scenario(
+ dimension_rows=tuple(
+ (key, draw(labels), draw(buckets)) for key in range(n_dims)
+ ),
+ fact_rows=tuple(
+ draw(
+ st.lists(
+ st.tuples(
+ st.one_of(st.none(), st.integers(0, n_dims)),
+ values,
+ values,
+ ),
+ max_size=10,
+ ),
+ ),
+ ),
+ metrics=draw(
+ st.tuples(st.sampled_from(metric_pool), st.sampled_from(metric_pool)),
+ ),
+ op=draw(st.sampled_from(["+", "-", "*", "/"])),
+ join_type=draw(st.sampled_from(["left", "inner"])),
+ dimensions=draw(dimension_subsets),
+ filters=tuple(draw(st.lists(filters, max_size=2))),
+ )
+
+
+@st.composite
+def materialization_scenarios(draw):
+ scenario = draw(scenarios())
+ required = set(scenario.dimensions)
+ for filter_ in scenario.filters:
+ required.update(filter_columns(filter_))
+ grain = tuple(
+ column
+ for column in DIMENSION_COLUMNS
+ if column in required or draw(st.booleans())
+ )
+ return replace(scenario, preagg_grain=grain)
diff --git a/datajunction-server/tests/property/sql_roundtrip_test.py b/datajunction-server/tests/property/sql_roundtrip_test.py
new file mode 100644
index 000000000..4d3f4685e
--- /dev/null
+++ b/datajunction-server/tests/property/sql_roundtrip_test.py
@@ -0,0 +1,188 @@
+"""
+Property: DJ's parser and printer preserve what an expression means.
+
+Random expressions are parsed and printed by DJ, then the original and DJ's
+version are both evaluated on DuckDB over random rows. Any difference in the
+results means DJ changed the query's meaning.
+"""
+
+import re
+
+import duckdb
+import pytest
+from hypothesis import assume, given, settings
+from hypothesis import strategies as st
+
+from datajunction_server.sql.parsing.backends.antlr4 import parse
+from tests.property.budget import examples
+
+COLUMNS = ["a", "b", "c"]
+
+numeric_leaf = st.one_of(
+ st.sampled_from(COLUMNS),
+ st.integers(-5, 5).map(str),
+ st.just("NULL"),
+)
+
+
+def numeric_branches(children):
+ return st.one_of(
+ st.tuples(children, st.sampled_from(["+", "-", "*"]), children).map(
+ lambda t: f"{t[0]} {t[1]} {t[2]}",
+ ),
+ children.map(lambda e: f"- {e}" if e.startswith("-") else f"-{e}"),
+ children.map(lambda e: f"- {e}"),
+ children.map(lambda e: f"({e})"),
+ st.tuples(children, children).map(lambda t: f"COALESCE({t[0]}, {t[1]})"),
+ children.map(lambda e: f"ABS({e})"),
+ st.tuples(boolean_expr(children), children, children).map(
+ lambda t: f"CASE WHEN {t[0]} THEN {t[1]} ELSE {t[2]} END",
+ ),
+ children.map(lambda e: f"CAST({e} AS BIGINT)"),
+ )
+
+
+def boolean_expr(numeric):
+ comparison = st.tuples(
+ numeric,
+ st.sampled_from(["=", "<>", "<", "<=", ">", ">="]),
+ numeric,
+ ).map(lambda t: f"{t[0]} {t[1]} {t[2]}")
+ atoms = st.one_of(
+ comparison,
+ numeric.map(lambda e: f"{e} IS NULL"),
+ numeric.map(lambda e: f"{e} IS NOT NULL"),
+ st.tuples(numeric, numeric, numeric).map(
+ lambda t: f"{t[0]} BETWEEN {t[1]} AND {t[2]}",
+ ),
+ st.tuples(numeric, st.lists(numeric, min_size=1, max_size=3)).map(
+ lambda t: f"{t[0]} IN ({', '.join(t[1])})",
+ ),
+ )
+ return st.recursive(
+ atoms,
+ lambda inner: st.one_of(
+ st.tuples(inner, st.sampled_from(["AND", "OR"]), inner).map(
+ lambda t: f"{t[0]} {t[1]} {t[2]}",
+ ),
+ inner.map(lambda e: f"NOT {e}"),
+ inner.map(lambda e: f"({e})"),
+ ),
+ max_leaves=4,
+ )
+
+
+numeric_expr = st.recursive(numeric_leaf, numeric_branches, max_leaves=6)
+
+string_leaf = st.one_of(
+ st.just("s"),
+ st.sampled_from(["''", "'x'", "'it''s'", "'a%'", "'%_'", "'\\\\'"]),
+)
+string_expr = st.recursive(
+ string_leaf,
+ lambda inner: st.one_of(
+ st.tuples(inner, inner).map(lambda t: f"CONCAT({t[0]}, {t[1]})"),
+ inner.map(lambda e: f"UPPER({e})"),
+ st.tuples(boolean_expr(numeric_leaf), inner, inner).map(
+ lambda t: f"CASE WHEN {t[0]} THEN {t[1]} ELSE {t[2]} END",
+ ),
+ ),
+ max_leaves=4,
+)
+string_predicates = st.one_of(
+ st.tuples(string_expr, string_expr).map(lambda t: f"{t[0]} = {t[1]}"),
+ st.tuples(string_expr, string_expr).map(lambda t: f"{t[0]} LIKE {t[1]}"),
+ st.tuples(string_expr, string_expr).map(lambda t: f"{t[0]} NOT LIKE {t[1]}"),
+)
+multi_case = st.tuples(
+ st.lists(
+ st.tuples(boolean_expr(numeric_leaf), numeric_leaf),
+ min_size=1,
+ max_size=3,
+ ),
+ st.one_of(st.none(), numeric_leaf),
+).map(
+ lambda t: (
+ "CASE "
+ + " ".join(f"WHEN {w} THEN {v}" for w, v in t[0])
+ + (f" ELSE {t[1]}" if t[1] is not None else "")
+ + " END"
+ ),
+)
+
+expressions = st.one_of(
+ numeric_expr,
+ boolean_expr(numeric_expr),
+ string_expr,
+ string_predicates,
+ multi_case,
+)
+
+# DJ prints `-(-x)` as `--x`, which SQL reads as a comment
+# (known_issues.PRINTER_DROPS_PARENTHESES); skipped so it doesn't mask other
+# differences.
+KNOWN_DOUBLE_MINUS = "--"
+
+rows = st.lists(
+ st.tuples(
+ *[st.one_of(st.none(), st.integers(-3, 3))] * len(COLUMNS),
+ st.one_of(st.none(), st.sampled_from(["", "x", "it's", "a%", "\\", "X"])),
+ ),
+ min_size=1,
+ max_size=6,
+)
+
+
+def evaluate(conn, expression: str):
+ return conn.execute(f"SELECT {expression} FROM t ORDER BY rowid").fetchall()
+
+
+@pytest.fixture(scope="module")
+def conn():
+ with duckdb.connect(":memory:") as connection:
+ yield connection
+
+
+def dj_roundtrip(expression: str) -> str:
+ query = parse(f"SELECT {expression} FROM t")
+ return str(query.select.projection[0])
+
+
+@settings(max_examples=examples(1000))
+@given(expression=expressions, data=rows)
+def test_dj_printing_preserves_meaning(conn, expression, data):
+ conn.execute(
+ "CREATE OR REPLACE TABLE t (a INTEGER, b INTEGER, c INTEGER, s VARCHAR)",
+ )
+ conn.executemany("INSERT INTO t VALUES (?, ?, ?, ?)", data)
+ try:
+ expected = evaluate(conn, expression)
+ except duckdb.Error:
+ assume(False)
+
+ printed = dj_roundtrip(expression)
+ assume(KNOWN_DOUBLE_MINUS not in printed)
+ try:
+ actual = evaluate(conn, printed)
+ except duckdb.Error as exc:
+ raise AssertionError(
+ f"DJ's version no longer runs.\n original: {expression}\n"
+ f" DJ: {printed}\n error: {exc}",
+ ) from exc
+ assert actual == expected, (
+ f"\n original: {expression}\n DJ: {printed}\n"
+ f" expected: {expected}\n got: {actual}"
+ )
+
+
+def without_negative_zero(sql: str) -> str:
+ # DJ reads the literal `-0` as 0; the value is the same either way.
+ return re.sub(r"(?=3.11, <3.14"
resolution-markers = [
"python_full_version >= '3.13'",
@@ -926,6 +926,7 @@ test = [
{ name = "gevent" },
{ name = "greenlet" },
{ name = "httpx" },
+ { name = "hypothesis" },
{ name = "mcp" },
{ name = "pre-commit" },
{ name = "pytest" },
@@ -1001,6 +1002,7 @@ test = [
{ name = "gevent", specifier = ">=24.2.1" },
{ name = "greenlet", specifier = ">=3.0.3" },
{ name = "httpx", specifier = ">=0.27.0" },
+ { name = "hypothesis", specifier = ">=6.168.1" },
{ name = "mcp", specifier = ">=1.0.0" },
{ name = "pre-commit", specifier = ">=3.2.2" },
{ name = "pytest", specifier = ">=7.3.0" },
@@ -1571,6 +1573,79 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/d2/fd/6668e5aec43ab844de6fc74927e155a3b37bf40d7c3790e49fc0406b6578/httpx_sse-0.4.3-py3-none-any.whl", hash = "sha256:0ac1c9fe3c0afad2e0ebb25a934a59f4c7823b60792691f779fad2c5568830fc", size = 8960, upload-time = "2025-10-10T21:48:21.158Z" },
]
+[[package]]
+name = "hypothesis"
+version = "6.168.1"
+source = { registry = "https://pypi.org/simple" }
+dependencies = [
+ { name = "sortedcontainers" },
+]
+sdist = { url = "https://files.pythonhosted.org/packages/aa/31/b54cd138ee36a32d0047d40fe7bdcb63cefd57ce098a26024161332d15fd/hypothesis-6.168.1.tar.gz", hash = "sha256:fd8acb5c67f260dfbdcc1b2dc95ff48b2b3f90708ae595c76f591bbf3f6f8eeb", size = 510823, upload-time = "2026-09-23T04:36:04.09Z" }
+wheels = [
+ { url = "https://files.pythonhosted.org/packages/48/95/3091b9ad24baa9666d49feea433a8c4060ad47245f0d529d736af3672ddd/hypothesis-6.168.1-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:5d9cc2d2a4779c92788389564fc6760df7c2909185f6431378133dea6933194f", size = 791345, upload-time = "2026-09-23T04:35:39.379Z" },
+ { url = "https://files.pythonhosted.org/packages/c5/b3/b954ffa69f91d111f801030c28d6721f422222af245a846571fc87874242/hypothesis-6.168.1-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d0e1ec15d32af11da6784d2506de0c901e4c1abf97a3cdcfc144225457783f4e", size = 787104, upload-time = "2026-09-23T04:35:21.799Z" },
+ { url = "https://files.pythonhosted.org/packages/e5/dc/732df08f570d845a480d3729bfb46beccc441bfabe2804bb3a458649be73/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4cc38519a57c4b919b2c59714a67501d20b46b0c96a0078cbdc9f86e728efce3", size = 1123848, upload-time = "2026-09-23T04:33:48.857Z" },
+ { url = "https://files.pythonhosted.org/packages/88/06/e1e937df42243267ba25ede5e9b29921186f290f6c0515729481cc41726e/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:7e229e937e761830c7fa6437fb12cb801616a1737af0898abf5edba57d2ad691", size = 1147698, upload-time = "2026-09-23T04:35:33.439Z" },
+ { url = "https://files.pythonhosted.org/packages/be/d1/3b41c008dc2a49e43bc4893dcd27ea9d9fef64fef50382ca5f7107680b1f/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:569ba21e5543840895d17633cc6716a9578ac83b1193a2978b3f24cc16b93085", size = 1149280, upload-time = "2026-09-23T04:35:45.514Z" },
+ { url = "https://files.pythonhosted.org/packages/71/ea/d54781058fe6a487b81ec6c636f3806fff6a417b48cf3fcd2b003d80f3ce/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:0a232665716607c3c50c9a3aeeac5929be1d4e58703f5e7cb51db7099445a8c6", size = 1191693, upload-time = "2026-09-23T04:35:57.896Z" },
+ { url = "https://files.pythonhosted.org/packages/4f/f2/43bdd07450978611d25989ffdcbe1579acd46c1500a484f8838359fe7880/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9022fe330d8839259da9b496573c4b83ea248bf32029eba23ff589f3c25754bf", size = 1169725, upload-time = "2026-09-23T04:34:32.59Z" },
+ { url = "https://files.pythonhosted.org/packages/d1/72/fd4b657cba66940f208bc0eccaceaa777553009d26e86068ff3ab5e55b0b/hypothesis-6.168.1-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:69be660b831d55ab35619b30abd311a6cb8cd0068a1192210d194a776b211094", size = 1129181, upload-time = "2026-09-23T04:35:27.495Z" },
+ { url = "https://files.pythonhosted.org/packages/d6/3f/516a29d7a2bd31e28972de635350eab086c94faa07afcdd3ad0f06e70db8/hypothesis-6.168.1-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:42e2cccca80ee4ffe8f66edf5b2adb2a16b8cd3c5110b3a774d70f2fc75a42aa", size = 1160162, upload-time = "2026-09-23T04:35:29.333Z" },
+ { url = "https://files.pythonhosted.org/packages/54/09/e2c7e32f281b31d4c19cc2fa935bfdde5b31ad4848a485e5ebb8de2389da/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:11cbd2a2539194f0960de4ed9f980882e3a3f55c476afcc6f4cc342eec5a07d2", size = 1299714, upload-time = "2026-09-23T04:35:12.529Z" },
+ { url = "https://files.pythonhosted.org/packages/ba/bc/85850bdb1fc35d0fa9e299ca3b2e354f403959666c7ef4d0bfda9c308eee/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:aa1620111616c660a1db03347b1790b5914ef30f86c6e8ddd6c765785291f0b4", size = 1425335, upload-time = "2026-09-23T04:35:23.557Z" },
+ { url = "https://files.pythonhosted.org/packages/40/c0/66bdb24f32d32aa5636655b6037af07962118b89fae16ee74bc9a1bb982e/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:a07c8665c95bb856f99ecd60b6f50d94dcf8157415387c32dcd5bb5b5061d032", size = 1376916, upload-time = "2026-09-23T04:34:59.742Z" },
+ { url = "https://files.pythonhosted.org/packages/31/f0/7ad0b17bda81c473cbc539009148f3e0950880c4618028691bcf62eeb8d2/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:8a50b6335d5fa2312dd9f5d50faeb953e61475b301d2c073cd7a4c5efd7e6171", size = 1280975, upload-time = "2026-09-23T04:34:21.966Z" },
+ { url = "https://files.pythonhosted.org/packages/e2/01/8d04d1eecf03a6097e26400e8393eff7a81fd52c877a664ee4e97adf7880/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:4ccb0276db8ad197a63dff25c4868e11123346347df6900625d2b1f794ce7c19", size = 1300238, upload-time = "2026-09-23T04:34:52.825Z" },
+ { url = "https://files.pythonhosted.org/packages/f5/74/eda0c70b3845789c713fe2ac72d3d4586da83f38a34f1f7aefbda0b5b01e/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:85ca7788574e06d0591d51e19ecbb1d09bc82d50c6eae5de0fe8317590b1e138", size = 1336088, upload-time = "2026-09-23T04:33:42.002Z" },
+ { url = "https://files.pythonhosted.org/packages/80/c9/43d8528e6d43eebb7688d36e530d1fc7763809b8f923b62cc8b4d95f99ff/hypothesis-6.168.1-cp310-abi3-win32.whl", hash = "sha256:77378e04aa47a22a615af80f42970dc44782f02b8a82ac105b619a2cc7f5eee2", size = 678004, upload-time = "2026-09-23T04:34:29.564Z" },
+ { url = "https://files.pythonhosted.org/packages/8c/8b/f7d4de5b57972f7f68a6c7fb5698f60425589eae6fcb175f285f2525e8c0/hypothesis-6.168.1-cp310-abi3-win_amd64.whl", hash = "sha256:f94d3dbc31112f8fbcb3af19a662ee5e90e380e6ae676357903caa14a1604763", size = 684722, upload-time = "2026-09-23T04:34:41.279Z" },
+ { url = "https://files.pythonhosted.org/packages/a5/fd/528eb221a82da0aa477e34c7af6ed2b4a5109b326638d4418b9e49435e96/hypothesis-6.168.1-cp310-abi3-win_arm64.whl", hash = "sha256:9bd7e8c6e2c3f6a5befee6b03ebd2170da2e50658075c6decad184777c72a726", size = 682711, upload-time = "2026-09-23T04:33:46.085Z" },
+ { url = "https://files.pythonhosted.org/packages/6d/1b/620484ba32fbb655ccbd02f3c510e7a961e840a466c3672ef0b877ab8716/hypothesis-6.168.1-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:9fd235c4ccfefe18fd97c181bdf247187cb3d4fdc8800a4d623322234ce4cfd3", size = 792049, upload-time = "2026-09-23T04:34:28.21Z" },
+ { url = "https://files.pythonhosted.org/packages/03/df/2ad250c2c4ee4d5f43d6d9bc5d852e7b81c771e4c8946212931b8cafe849/hypothesis-6.168.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:0722212cf825d001a0d3b4db9303cc31af467f6814b69acab91d91c3f629c95a", size = 787932, upload-time = "2026-09-23T04:35:20.079Z" },
+ { url = "https://files.pythonhosted.org/packages/24/69/4f8e8bf1d5ad81352a05aa0835baef82e691e3e43933bafe6d6f2c171763/hypothesis-6.168.1-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:94a1053575f133ff6978876fd4b4c2a153c1111f1b5f14d6cfb355c306fc4cb5", size = 1124000, upload-time = "2026-09-23T04:34:14.106Z" },
+ { url = "https://files.pythonhosted.org/packages/b6/e7/b72a34e0c6b434d3ad2b2e0040fa88d06a15a7880c59bddacf1e27d4902a/hypothesis-6.168.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6d2c74d496304574cfed537e4f657620be7fee3e287381c0aa4b481211e6ec92", size = 1170172, upload-time = "2026-09-23T04:34:30.93Z" },
+ { url = "https://files.pythonhosted.org/packages/b6/6c/afe1b73b58479ba0d75e5408259978d580e96fee812eb4ea2bb83a76f7ff/hypothesis-6.168.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:cf4cc6819ba9689effba38dce98a3025554669f76be07e47cc1002e93e8c4918", size = 1300163, upload-time = "2026-09-23T04:35:53.549Z" },
+ { url = "https://files.pythonhosted.org/packages/b0/3d/aad54ffe7942c06da79c719fc98daa1dffdd8ac2d384ca64cc3611b1a833/hypothesis-6.168.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:7888e0876e1d93fb92a6a4869834b0d0de18f07f6600edbf3d1cad16cf089357", size = 1336274, upload-time = "2026-09-23T04:35:10.655Z" },
+ { url = "https://files.pythonhosted.org/packages/0f/cb/8d2a3474e7921cfb5e59f933ac6fe1ad83b95cc98a871af8f0a68dcf946e/hypothesis-6.168.1-cp311-cp311-win_amd64.whl", hash = "sha256:4aa312b6447743a72283698292ea0e1c4d684279b38dc7f270729c4efda6c27a", size = 684479, upload-time = "2026-09-23T04:33:59.307Z" },
+ { url = "https://files.pythonhosted.org/packages/7e/6f/9d5ae55d81e10f0fc46868c19ab9d468e9959e82ff8cc8c26990b16114f3/hypothesis-6.168.1-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:dfe14a88b1ed03ce47333d800cc37949da73feb0cd9767e7ef076d2202768928", size = 793124, upload-time = "2026-09-23T04:33:52.589Z" },
+ { url = "https://files.pythonhosted.org/packages/14/11/54c19d259d7f92176aa2199c90f7951e4e396a26898a596dc875ccb9c389/hypothesis-6.168.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:8acd000c3bf233cdcd28c1ff03f1bf56cb03821b1c742d00d57978b20fc6e6f8", size = 784674, upload-time = "2026-09-23T04:34:38.137Z" },
+ { url = "https://files.pythonhosted.org/packages/5f/ae/f80a9810f88bd654ab0d0ed0652c274e4c0384e22d4f9442b6753b9200a6/hypothesis-6.168.1-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:29280d4743546ddcf74644f9acdc5726972cba9182699a86b7e77f3f8f6e5dde", size = 1122860, upload-time = "2026-09-23T04:33:51.375Z" },
+ { url = "https://files.pythonhosted.org/packages/f4/5d/aa5b7ffd57ec475a9252c55985c07a366823c3f85267d44648e94c1ec321/hypothesis-6.168.1-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:bd9a488d46b8c7da3103058c7569376a3a08994aee9a7d6f1d160d3d3b49b82d", size = 1168922, upload-time = "2026-09-23T04:34:44.567Z" },
+ { url = "https://files.pythonhosted.org/packages/65/10/4862448192d4c9cc61a3b8a620730800f56b1e83b0bfda63dbc728526eb3/hypothesis-6.168.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:451ee6c42955887a9e1850c53813e31182eab1663e4d3dacea4e535dedc45b5d", size = 1298568, upload-time = "2026-09-23T04:35:31.191Z" },
+ { url = "https://files.pythonhosted.org/packages/15/ad/55e73fda9afbfcf3336b2eee564833bc9a4fee1f7a32fe5f9d183742da39/hypothesis-6.168.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:bc2c036f4a9534af64977e38b2976295136f83417fdb95bed816a52cc3023497", size = 1335091, upload-time = "2026-09-23T04:34:23.42Z" },
+ { url = "https://files.pythonhosted.org/packages/ef/99/b6340fa21a89b0088ed023fe984109e1c47a07dfd741fd6d8e17a9e2795f/hypothesis-6.168.1-cp312-cp312-win_amd64.whl", hash = "sha256:d603d7c96d030a63701b6ad9bff0bc4870e4c226335b3db509f1eed1f7fc061b", size = 682011, upload-time = "2026-09-23T04:34:34.565Z" },
+ { url = "https://files.pythonhosted.org/packages/f5/23/093b9dc768e89d2cec74d29115f4004fe1be64218bac65f8d817d727d30b/hypothesis-6.168.1-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:aebf7e9d7ba9f920e154abc65851d4affab497daeaea285474bed93840a983ea", size = 793062, upload-time = "2026-09-23T04:33:58.177Z" },
+ { url = "https://files.pythonhosted.org/packages/df/d7/f191686b44b3f892ae1404e238a55260152a645a2deafe8b2949c7e90fe9/hypothesis-6.168.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:b26616604cac458dd83e75900fefd9190f1074e67d5d9dc4d9e321d30f295eca", size = 784518, upload-time = "2026-09-23T04:34:02.012Z" },
+ { url = "https://files.pythonhosted.org/packages/10/25/b7a5dca1911c9012a564d0b7aaeb44e393ade9424a695dea179ff32d79fe/hypothesis-6.168.1-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:cb6d89cff2edcd45e5af957c7b6a71ae7887419561f7b9c7055e1c50dea41017", size = 1122844, upload-time = "2026-09-23T04:35:06.897Z" },
+ { url = "https://files.pythonhosted.org/packages/ea/60/ddacd0d244e6719ea9a9d322d4d4b6b054fd27ada94ec7aa7dce7ef11297/hypothesis-6.168.1-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:2a4ce6310ec21000446ae75cdc6667b81d4ee0f2d226a0dfaf8baca42dd151d0", size = 1168779, upload-time = "2026-09-23T04:33:56.885Z" },
+ { url = "https://files.pythonhosted.org/packages/dc/07/343bab6676a41f38cb7c5912ab3ca059258bf40e41c181d1b77cc7251801/hypothesis-6.168.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:a22c6e730c327d22a4d61b53f66639cdf6f0fe4a5a1b3595c36533e5b8127169", size = 1298434, upload-time = "2026-09-23T04:34:48.576Z" },
+ { url = "https://files.pythonhosted.org/packages/7d/bd/4b72fafa5f46a455024a9242b62505b1a53fad95ace162e831c4ffb233b9/hypothesis-6.168.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:dcd66493fe180d38641b02dd0456cc1e429655f8940dd1396ab874808f16dd34", size = 1334975, upload-time = "2026-09-23T04:35:01.531Z" },
+ { url = "https://files.pythonhosted.org/packages/3f/ae/48643915cda400ca2dccb9bcdc363690b13bcd5c179172bc6450c3d76ce8/hypothesis-6.168.1-cp313-cp313-win_amd64.whl", hash = "sha256:a8a96f2e9fa2a57f95865aaf092943775e7ea96f72535371b2be20fe5464298c", size = 681978, upload-time = "2026-09-23T04:36:00.056Z" },
+ { url = "https://files.pythonhosted.org/packages/54/e9/45595951f8bc9bd03f032325e7b176ca990a0c86bf2e09d481e8ee5a5cab/hypothesis-6.168.1-cp315-abi3.abi3t-macosx_10_12_x86_64.whl", hash = "sha256:e9510e33467d586a7700d56b9a6ee65b44a06832c1cab3d270369d7c53381276", size = 791046, upload-time = "2026-09-23T04:35:55.724Z" },
+ { url = "https://files.pythonhosted.org/packages/8d/3e/718431d0ab72c4c07454f71040b4de298a63e765f20088557dc1bef0a48b/hypothesis-6.168.1-cp315-abi3.abi3t-macosx_11_0_arm64.whl", hash = "sha256:99970e3a75f11dde0414d3d6724bd8b7a2bd1deeb9a509ec9236c91d41a5cff0", size = 782982, upload-time = "2026-09-23T04:34:06.017Z" },
+ { url = "https://files.pythonhosted.org/packages/d7/05/c60db6df1456a70a3c589e42acaa446c22f56ed342c1b8dd04b8e49df8fd/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a5ea78773de5bcc25266e37eb086472e90a85fc81b8d4a9ca8afc15858ee15fb", size = 1120961, upload-time = "2026-09-23T04:33:43.494Z" },
+ { url = "https://files.pythonhosted.org/packages/8b/fe/d943250395750ecc3b4ec95b1ab3c8119f6cbad08c1ba9169ffee7d89049/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:13f502ce2423e46dfcd2af38950a86ecf97084d94e721f7f13dd5a70afb36e0a", size = 1143877, upload-time = "2026-09-23T04:34:56.065Z" },
+ { url = "https://files.pythonhosted.org/packages/d3/9d/8c1f8e11f20fd81ce40caed7ad6d6a9776e24775070e87f3aa9ba3683272/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:69b4bcaa8b0e753fc8921011d5d10c8a7c3caf814c6a968bc140a08917973fdc", size = 1146503, upload-time = "2026-09-23T04:34:04.791Z" },
+ { url = "https://files.pythonhosted.org/packages/71/d8/ac2e9fa653d2f05a13724837007e857d376d81fdf728543ab85f6c7f7350/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:7ed811908d38ca8e05c20ad25ecb61edc1e20a519667d4b668be49dd09e6bf3e", size = 1189216, upload-time = "2026-09-23T04:34:36.271Z" },
+ { url = "https://files.pythonhosted.org/packages/2b/7b/640952448781c7df5098880e2e2d6e7702b3a1c0badd4c0ea5fca6d6ab53/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9239c868418088d58ae18d041f9f424bd7a75dd143f209c35fde38770b65f74f", size = 1166919, upload-time = "2026-09-23T04:35:16.498Z" },
+ { url = "https://files.pythonhosted.org/packages/b6/de/0d46b330a28f7614850fb1a0564cf88af7c0b29cbd791e9d04ade7f66f13/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_31_riscv64.whl", hash = "sha256:5cac279798ecc92d3301ec4421c0458e22627f8a79fdde8a6c98ab1d4b8811f6", size = 1126652, upload-time = "2026-09-23T04:35:49.432Z" },
+ { url = "https://files.pythonhosted.org/packages/92/36/6adeca1724ed4966095406ffe7308fcff338cf9f5b7443d5ea34a01ab6e1/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:6bc660dc14d633641e22e407417d35bf2925331273ed71bc6ab5536dfca5c72d", size = 1155687, upload-time = "2026-09-23T04:35:18.234Z" },
+ { url = "https://files.pythonhosted.org/packages/bd/00/e9b7c7e4a0a75d1621be95a126999352e01a16d68b0a59b396695c7b4991/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_aarch64.whl", hash = "sha256:157a29e4d6bcb77ae732cb83d6546beef47265ecac88408fdd5d5b2bd325685d", size = 1296471, upload-time = "2026-09-23T04:34:12.763Z" },
+ { url = "https://files.pythonhosted.org/packages/8f/d6/52cb320e9e07cc67f718506617b5900344370394e39f4affada317aab313/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_armv7l.whl", hash = "sha256:b125aac89594ee2a76687b724f057b2423ea83dfee6b8ff59f54289d6eef19a1", size = 1421838, upload-time = "2026-09-23T04:34:20.451Z" },
+ { url = "https://files.pythonhosted.org/packages/9b/93/969c33a87cd2865beffebe491294b8115e7e7421c6840f79a67486535bfc/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_i686.whl", hash = "sha256:4b960c9d31b85dbc6a25888897c02bbea5f498844932841612925184fc78d9eb", size = 1373994, upload-time = "2026-09-23T04:34:00.608Z" },
+ { url = "https://files.pythonhosted.org/packages/34/c9/5fe4ade4bd23baa2e4cd27668bb8ac10e840ce19fbcfc74116ea529cbaa7/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_ppc64le.whl", hash = "sha256:d4d76f35874471fe9b8e1bb696fb532363ef3d9588161a06d30520f638f15a39", size = 1278216, upload-time = "2026-09-23T04:34:07.253Z" },
+ { url = "https://files.pythonhosted.org/packages/a1/17/30b9a1ad975e90700dea74ce8b08ab07b9e58c4d83918422bc8841942c1e/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_riscv64.whl", hash = "sha256:d5147c5f7497d109eed0267c498adaf5cad6e4beeaca6fad6057f2007bae70a7", size = 1297647, upload-time = "2026-09-23T04:35:51.491Z" },
+ { url = "https://files.pythonhosted.org/packages/e0/2f/233a9d5adafad6e187ab8f4d9275592cf120b56e63ea1cfb02213c01021e/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_x86_64.whl", hash = "sha256:a909f89758e184c4f86418598de297a2ced5b922979c0f98c5d73ff6bafe38e5", size = 1333790, upload-time = "2026-09-23T04:34:42.828Z" },
+ { url = "https://files.pythonhosted.org/packages/10/d7/4c9d4d597f70316d781cd99b600adae090a6f5d675487196e10510a23afe/hypothesis-6.168.1-cp315-abi3.abi3t-win32.whl", hash = "sha256:66b9fb6f02840f8ffa2a4e6359aff6ecc1b2c6630c8a2066771044fa22956c7a", size = 675206, upload-time = "2026-09-23T04:35:14.592Z" },
+ { url = "https://files.pythonhosted.org/packages/b4/a8/f9fc0c0dbb9e80238fc7145cb4ce52e0897ea417a23217022911855bccda/hypothesis-6.168.1-cp315-abi3.abi3t-win_amd64.whl", hash = "sha256:9246eaa80044c33a182be661cbed7e6a7314c61359ef501ea9e9efd90d50ca3b", size = 681503, upload-time = "2026-09-23T04:34:11.223Z" },
+ { url = "https://files.pythonhosted.org/packages/11/b4/962ae350c10860be37c9abf98c0ef1ef23e4baca3ee7ad77208941e370e5/hypothesis-6.168.1-cp315-abi3.abi3t-win_arm64.whl", hash = "sha256:52e0c804cbf697d4760400c84444a91cbd5dfe7ae0ef1bb3228f74fb5547eb9e", size = 679199, upload-time = "2026-09-23T04:34:03.472Z" },
+ { url = "https://files.pythonhosted.org/packages/aa/c5/2bfdcc601b08eb9611bc524d18ca0a8cdbbdd7bd07bc68c38e0196a00668/hypothesis-6.168.1-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:91e3700d9c35e184dd253cff2f151f890557242bede75a2535901e9062d42975", size = 792943, upload-time = "2026-09-23T04:35:25.308Z" },
+ { url = "https://files.pythonhosted.org/packages/4b/52/36ae5128d9102d73dc578cd7b241c33f5483bd56e05799a2d06043a1ce7d/hypothesis-6.168.1-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:c19bc5da324a5899527ea749522e9ceb47f1eb54772869673abe59ae0b703aef", size = 788802, upload-time = "2026-09-23T04:34:39.655Z" },
+ { url = "https://files.pythonhosted.org/packages/23/34/698fc3481757503dc60f07547c29098dce97ac035a1ee134d66d88814281/hypothesis-6.168.1-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4a4c9e46b2f349658d3013a357e1fbf114bc08abced7ee8e91cd6e0bc718b195", size = 1124766, upload-time = "2026-09-23T04:34:57.869Z" },
+ { url = "https://files.pythonhosted.org/packages/02/c9/1f871666106d90048027fc6b309ec505f0fb44e4519add73ddaad98bcd9a/hypothesis-6.168.1-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:95a0beee16fcbc0516951f2ef3ef1a479ad11d9726da49840139303b36ebc1d0", size = 1171766, upload-time = "2026-09-23T04:34:54.476Z" },
+ { url = "https://files.pythonhosted.org/packages/17/44/eff662526259ac4447dde4abe4fc285d0e8e5ff5345d35af836c6f0f12cc/hypothesis-6.168.1-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:4a8beb513c066fcb187b885fdfa93da4006f223d50d605b687bfb7e60aac33e9", size = 685449, upload-time = "2026-09-23T04:33:44.837Z" },
+]
+
[[package]]
name = "identify"
version = "2.6.16"