diff --git a/CONTRIBUTING.rst b/CONTRIBUTING.rst index 6ce8b6451..f6b820ede 100644 --- a/CONTRIBUTING.rst +++ b/CONTRIBUTING.rst @@ -322,6 +322,44 @@ If you prefer to use tox, these flags all work the same way. tox tests/sql/parsing/queries/tpcds/test_tpcds.py::test_parsing_sparksql_tpcds_queries -- --tpcds +Property-Based Tests +-------------------- + +``datajunction-server/tests/property/`` holds `Hypothesis `_ tests. They generate +metric definitions, small fact and dimension tables, requests, and SQL expressions to check correctness rules. Most +execute DJ's SQL on DuckDB and compare it with a direct calculation or query: + +- ``metric_decomposition_test.py``: a metric rolled up from its components equals the metric computed directly. +- ``sql_roundtrip_test.py``: parsing and printing SQL keeps its meaning. +- ``ast_printing_test.py``: AST-built expressions with unambiguous operator precedence print as equivalent SQL. +- ``random_graph_test.py``: SQL for a random graph and request matches a plain query over the raw tables. +- ``materialization_test.py``: a request served from a pre-aggregation matches the same request computed from raw + tables. Both paths use DJ-generated SQL; ``random_graph_test.py`` supplies the independent raw-table comparison. + +``tests/property/scenario.py`` defines the immutable case, generated metric and filter specifications, and strategies. +Each metric's raw-table reference expression is written separately from its DJ definition. ``graph.py`` installs a +case through DJ's API and runs the direct DuckDB reference query. Add new case variants in ``scenario.py`` so the same +specifications can be reused across properties. + +They run with the rest of the server suite. ``DJ_PBT_PROFILE`` selects ``dev`` (the default locally, random with +shrinking), ``ci`` (the default when ``CI`` is set, a fixed sequence without shrinking), or ``nightly`` (ten times the +examples, random with shrinking). CI runs only ``ci``, which replays the same examples every time, so it does not +search for new bugs. ``nightly`` is not scheduled anywhere; run it by hand for a deeper search. The API-backed properties need the same Postgres test setup as the server suite; +locally, the test fixtures start it through Docker. + +.. code-block:: sh + + cd datajunction-server + uv run pytest tests/property -n auto + DJ_PBT_PROFILE=nightly uv run pytest tests/property -n auto + +In ``dev`` and ``nightly``, Hypothesis shrinks failures to a smaller example. The ``ci`` profile reports its failing +example without shrinking. The ``@reproduce_failure`` decorator in failure output can replay an example exactly. + +Bugs found this way and not yet fixed are listed in ``tests/property/known_issues.py``. The generated tests exclude +affected inputs or metric families, and each listed bug has a small repro marked ``xfail(strict=True)``. Fixing a bug +makes its repro pass, which fails the run; then remove the marker and its corresponding exclusion. + Enabling ``pdb`` When Running Tests ----------------------------------- @@ -432,4 +470,4 @@ The easiest way to fix it is to reset your database state using these commands ( root@...:/code# alembic upgrade head ... -After this, the `docker compose up` command should start the db_migration agent without problems. \ No newline at end of file +After this, the `docker compose up` command should start the db_migration agent without problems. diff --git a/datajunction-server/pyproject.toml b/datajunction-server/pyproject.toml index 45df7edeb..cc2f43f8a 100644 --- a/datajunction-server/pyproject.toml +++ b/datajunction-server/pyproject.toml @@ -161,6 +161,7 @@ testpaths = [ ] norecursedirs = [ "tests/helpers", + ".hypothesis", ] [tool.ruff.lint] @@ -188,4 +189,5 @@ test = [ "sqlparse<1.0.0,>=0.4.3", "asgi-lifespan>=2", "mcp>=1.0.0", + "hypothesis>=6.168.1", ] diff --git a/datajunction-server/tests/property/__init__.py b/datajunction-server/tests/property/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/datajunction-server/tests/property/ast_printing_test.py b/datajunction-server/tests/property/ast_printing_test.py new file mode 100644 index 000000000..6806437f9 --- /dev/null +++ b/datajunction-server/tests/property/ast_printing_test.py @@ -0,0 +1,203 @@ +""" +Property: an expression tree built in code prints as SQL that means the same +thing. + +DJ assembles SQL by constructing AST nodes (metric combiners, combined filters, +derived-metric substitution). This property covers expression trees whose +operators have unambiguous precedence while the known nested-parentheses bug +has a separate, fixed repro. Each tree is rendered by DJ and by a reference +renderer that parenthesizes every sub-expression, then evaluated on DuckDB. +""" + +import duckdb +import pytest +from hypothesis import given, settings +from hypothesis import strategies as st + +from datajunction_server.sql.parsing import ast +from tests.property.budget import examples +from tests.property.comparison import PropertyMismatch +from tests.property.known_issues import PRINTER_DROPS_PARENTHESES + + +class Expr: + """A generated expression: its DJ AST and its fully parenthesized SQL.""" + + def __init__(self, build, reference: str): + self.build = build # fresh AST on each call; nodes carry parent links + self.reference = reference + + def __repr__(self) -> str: + return self.reference + + +def column(name): + return Expr(lambda: ast.Column(ast.Name(name)), name) + + +def number(value): + return Expr(lambda: ast.Number(value), f"({value})" if value < 0 else str(value)) + + +NULL = Expr(lambda: ast.Null(), "NULL") + + +def binary(op: ast.BinaryOpKind, left: Expr, right: Expr) -> Expr: + return Expr( + lambda: ast.BinaryOp(op=op, left=left.build(), right=right.build()), + f"({left.reference} {op.value} {right.reference})", + ) + + +def negate(e: Expr) -> Expr: + return Expr( + lambda: ast.ArithmeticUnaryOp( + op=ast.ArithmeticUnaryOpKind.Minus, + expr=e.build(), + ), + f"(-{e.reference})", + ) + + +def not_(e: Expr) -> Expr: + return Expr( + lambda: ast.UnaryOp(op=ast.UnaryOpKind.Not, expr=e.build()), + f"(NOT {e.reference})", + ) + + +def is_null(e: Expr, negated: bool) -> Expr: + return Expr( + lambda: ast.IsNull(expr=e.build(), negated=negated), + f"({e.reference} IS {'NOT ' if negated else ''}NULL)", + ) + + +def between(e: Expr, low: Expr, high: Expr, negated: bool) -> Expr: + return Expr( + lambda: ast.Between( + expr=e.build(), + low=low.build(), + high=high.build(), + negated=negated, + ), + f"({e.reference} {'NOT ' if negated else ''}BETWEEN " + f"{low.reference} AND {high.reference})", + ) + + +def function(name: str, *args: Expr) -> Expr: + return Expr( + lambda: ast.Function(ast.Name(name), args=[a.build() for a in args]), + f"{name}({', '.join(a.reference for a in args)})", + ) + + +def case(condition: Expr, result: Expr, otherwise: Expr) -> Expr: + return Expr( + lambda: ast.Case( + conditions=[condition.build()], + results=[result.build()], + else_result=otherwise.build(), + ), + f"(CASE WHEN {condition.reference} THEN {result.reference} " + f"ELSE {otherwise.reference} END)", + ) + + +ARITHMETIC = [ast.BinaryOpKind.Plus, ast.BinaryOpKind.Minus, ast.BinaryOpKind.Multiply] +COMPARISON = [ + ast.BinaryOpKind.Eq, + ast.BinaryOpKind.NotEq, + ast.BinaryOpKind.Lt, + ast.BinaryOpKind.GtEq, +] +LOGICAL = [ast.BinaryOpKind.And, ast.BinaryOpKind.Or] + +numeric_leaves = st.one_of( + st.sampled_from(["a", "b", "c"]).map(column), + st.integers(0, 3).map(number), + st.just(NULL), +) + + +base_comparison = st.builds( + binary, + st.sampled_from(COMPARISON), + numeric_leaves, + numeric_leaves, +) +numeric_atoms = st.one_of( + numeric_leaves, + st.builds(function, st.just("COALESCE"), numeric_leaves, numeric_leaves), + st.builds(case, base_comparison, numeric_leaves, numeric_leaves), +) +boolean_atoms = st.one_of( + st.builds(binary, st.sampled_from(COMPARISON), numeric_atoms, numeric_atoms), + st.builds(is_null, numeric_atoms, st.booleans()), + st.builds(between, numeric_atoms, numeric_leaves, numeric_leaves, st.booleans()), +) +expressions = st.one_of( + numeric_atoms, + st.builds(binary, st.sampled_from(ARITHMETIC), numeric_atoms, numeric_atoms), + numeric_atoms.map(negate), + boolean_atoms, + st.builds(binary, st.sampled_from(LOGICAL), boolean_atoms, boolean_atoms), + boolean_atoms.map(not_), +) + + +rows = st.lists( + st.tuples(*[st.one_of(st.none(), st.integers(-3, 3))] * 3), + min_size=1, + max_size=6, +) + + +def evaluate(conn, sql: str): + return conn.execute(f"SELECT {sql} FROM t ORDER BY rowid").fetchall() + + +@pytest.fixture(scope="module") +def conn(): + with duckdb.connect(":memory:") as connection: + yield connection + + +@settings(max_examples=examples(500)) +@given(expression=expressions, data=rows) +def test_printed_tree_means_the_same_as_the_tree(conn, expression, data): + conn.execute("CREATE OR REPLACE TABLE t (a INTEGER, b INTEGER, c INTEGER)") + conn.executemany("INSERT INTO t VALUES (?, ?, ?)", data) + expected = evaluate(conn, expression.reference) + + printed = str(expression.build()) + try: + actual = evaluate(conn, printed) + except duckdb.Error as exc: + raise AssertionError( + f"DJ's SQL no longer runs.\n tree: {expression.reference}\n" + f" DJ: {printed}\n error: {exc}", + ) from exc + assert actual == expected, ( + f"\n tree: {expression.reference}\n DJ: {printed}\n" + f" expected: {expected}\n got: {actual}" + ) + + +@PRINTER_DROPS_PARENTHESES.xfail() +def test_known_issue_nested_negation_prints_as_comment(conn): + conn.execute("CREATE OR REPLACE TABLE t (a INTEGER, b INTEGER, c INTEGER)") + conn.execute("INSERT INTO t VALUES (1, 2, 3)") + expression = negate(negate(column("a"))) + expected = evaluate(conn, expression.reference) + printed = str(expression.build()) + try: + actual = evaluate(conn, printed) + except duckdb.Error as exc: + if printed.startswith("--"): + raise PropertyMismatch( + f"DJ printed {printed!r} for {expression!r}", + ) from exc + raise + assert actual == expected diff --git a/datajunction-server/tests/property/budget.py b/datajunction-server/tests/property/budget.py new file mode 100644 index 000000000..1a3810599 --- /dev/null +++ b/datajunction-server/tests/property/budget.py @@ -0,0 +1,19 @@ +""" +How many examples each property runs. + +Properties declare a budget for an ordinary PR run; `DJ_PBT_PROFILE=nightly` +multiplies it for longer searches. +""" + +import os + +SCALE = {"dev": 1, "ci": 1, "nightly": 10} + + +def profile() -> str: + default = "ci" if os.environ.get("CI") else "dev" + return os.environ.get("DJ_PBT_PROFILE", default) + + +def examples(n: int) -> int: + return n * SCALE[profile()] diff --git a/datajunction-server/tests/property/comparison.py b/datajunction-server/tests/property/comparison.py new file mode 100644 index 000000000..5127b1800 --- /dev/null +++ b/datajunction-server/tests/property/comparison.py @@ -0,0 +1,18 @@ +"""Value comparison and failure type for property-test correctness oracles.""" + +import math + + +class PropertyMismatch(AssertionError): + """The result of a DJ operation differs from the independent answer.""" + + +def same_value(actual, expected) -> bool: + """Compare numbers approximately, but keep SQL NULL and non-finite values distinct.""" + if actual is None or expected is None: + return actual is None and expected is None + if isinstance(actual, float) and math.isnan(actual): + return isinstance(expected, float) and math.isnan(expected) + if isinstance(expected, float) and math.isnan(expected): + return False + return math.isclose(actual, expected, rel_tol=1e-6, abs_tol=1e-6) diff --git a/datajunction-server/tests/property/comparison_test.py b/datajunction-server/tests/property/comparison_test.py new file mode 100644 index 000000000..6a98ca700 --- /dev/null +++ b/datajunction-server/tests/property/comparison_test.py @@ -0,0 +1,25 @@ +"""The oracle must not conflate distinct SQL result values.""" + +import math + +import pytest + +from tests.property.comparison import same_value + + +@pytest.mark.parametrize( + ("actual", "expected", "matches"), + [ + (None, None, True), + (None, math.nan, False), + (None, math.inf, False), + (math.nan, math.nan, True), + (math.nan, math.inf, False), + (math.inf, math.inf, True), + (math.inf, -math.inf, False), + (1.0, 1.0 + 1e-7, True), + (1.0, 1.1, False), + ], +) +def test_same_value(actual, expected, matches): + assert same_value(actual, expected) is matches diff --git a/datajunction-server/tests/property/conftest.py b/datajunction-server/tests/property/conftest.py new file mode 100644 index 000000000..51d5285dd --- /dev/null +++ b/datajunction-server/tests/property/conftest.py @@ -0,0 +1,34 @@ +""" +Hypothesis profiles for the property tests, chosen with `DJ_PBT_PROFILE`: + +- dev (default locally): random examples; failures are saved and replayed. +- ci (default when `CI` is set): a fixed sequence of examples and no shrinking, + so runs are repeatable and a failure is reported quickly. +- nightly: ten times the examples, random, with shrinking. +""" + +from hypothesis import HealthCheck, Phase, settings + +from tests.property.budget import profile + +COMMON = { + # Examples that create nodes through the API take well over the default + # 200ms, and vary with load. + "deadline": None, + "suppress_health_check": [ + HealthCheck.function_scoped_fixture, + HealthCheck.too_slow, + ], + "print_blob": True, +} + +settings.register_profile("dev", **COMMON) +settings.register_profile( + "ci", + derandomize=True, + database=None, + phases=[Phase.explicit, Phase.reuse, Phase.generate], + **COMMON, +) +settings.register_profile("nightly", **COMMON) +settings.load_profile(profile()) diff --git a/datajunction-server/tests/property/graph.py b/datajunction-server/tests/property/graph.py new file mode 100644 index 000000000..f0a254823 --- /dev/null +++ b/datajunction-server/tests/property/graph.py @@ -0,0 +1,236 @@ +""" +Random fact + dimension graphs, created in DJ through its API and mirrored as +tables in DuckDB, for properties that compare DJ's SQL against other answers. +""" + +import itertools + +from httpx import AsyncClient + +from tests.conftest import transpile_to_duckdb +from tests.property.comparison import PropertyMismatch, same_value +from tests.property.scenario import ( + DimensionColumn, + FilterSpec, + Scenario, + render_filter, +) + +counter = itertools.count() + + +def accepts_missing_dimension(conn, filters: tuple[FilterSpec, ...]) -> bool: + """Whether all filters accept a fact with no matching dimension row.""" + if not filters: + return True + + def missing_column(column: DimensionColumn) -> str: + return "CAST(NULL AS VARCHAR)" if column == "label" else "CAST(NULL AS INTEGER)" + + where = " AND ".join( + f"({render_filter(filter_, missing_column)})" for filter_ in filters + ) + return bool(conn.sql(f"SELECT COALESCE(({where}), FALSE)").fetchone()[0]) + + +def has_orphan_facts(scenario: Scenario) -> bool: + keys = {key for key, _, _ in scenario.dimension_rows} + return any(key not in keys for key, _, _ in scenario.fact_rows) + + +def by_group(rows, width: int) -> dict: + """Rows keyed by their first `width` columns (the requested dimensions).""" + grouped = {} + for row in rows: + key = row[:width] + assert key not in grouped, f"duplicate result group: {key}" + grouped[key] = row[width:] + return grouped + + +def assert_same_results(actual: dict, expected: dict, context: str) -> None: + wrong = { + key: (actual.get(key, ""), expected.get(key, "")) + for key in actual.keys() | expected.keys() + if key not in actual + or key not in expected + or len(actual[key]) != len(expected[key]) + or not all(same_value(a, e) for a, e in zip(actual[key], expected[key])) + } + if wrong: + raise PropertyMismatch(f"\n{context}\n(actual, expected) by group: {wrong}") + + +class Graph: + def __init__(self, n: int): + self.fact_table, self.dim_table = f"fact_{n}", f"dim_{n}" + self.fact = f"default.pbt{n}_fact" + self.dim_src = f"default.pbt{n}_dim_src" + self.dim = f"default.pbt{n}_dim" + self.prefix = f"default.pbt{n}_" + + def dimension(self, column: str) -> str: + return f"{self.dim}.{column}" + + +async def post(client: AsyncClient, path: str, payload: dict): + response = await client.post(path, json=payload) + assert response.status_code in (200, 201), (path, response.json()) + return response.json() + + +async def build_graph(client, conn, scenario: Scenario) -> Graph: + graph = Graph(next(counter)) + conn.execute('CREATE SCHEMA IF NOT EXISTS "default".pbt') + conn.execute( + f'CREATE TABLE "default".pbt.{graph.dim_table} ' + "(k INTEGER, label VARCHAR, bucket INTEGER)", + ) + conn.execute( + f'CREATE TABLE "default".pbt.{graph.fact_table} ' + "(k INTEGER, x INTEGER, y INTEGER)", + ) + conn.executemany( + f'INSERT INTO "default".pbt.{graph.dim_table} VALUES (?, ?, ?)', + scenario.dimension_rows, + ) + if scenario.fact_rows: + conn.executemany( + f'INSERT INTO "default".pbt.{graph.fact_table} VALUES (?, ?, ?)', + scenario.fact_rows, + ) + + def source(name, table, columns): + return { + "name": name, + "catalog": "default", + "schema_": "pbt", + "table": table, + "mode": "published", + "columns": [{"name": c, "type": t} for c, t in columns], + } + + await post( + client, + "/nodes/source/", + source( + graph.fact, + graph.fact_table, + [("k", "int"), ("x", "int"), ("y", "int")], + ), + ) + await post( + client, + "/nodes/source/", + source( + graph.dim_src, + graph.dim_table, + [("k", "int"), ("label", "string"), ("bucket", "int")], + ), + ) + await post( + client, + "/nodes/dimension/", + { + "name": graph.dim, + "query": f"SELECT k, label, bucket FROM {graph.dim_src}", + "primary_key": ["k"], + "mode": "published", + }, + ) + await post( + client, + f"/nodes/{graph.fact}/link", + { + "dimension_node": graph.dim, + "join_type": scenario.join_type, + "join_on": f"{graph.fact}.k = {graph.dim}.k", + }, + ) + return graph + + +async def create_metric(client, graph: Graph, name: str, query: str) -> str: + metric = graph.prefix + name + await post( + client, + "/nodes/metric/", + {"name": metric, "query": query, "mode": "published"}, + ) + return metric + + +async def metrics_sql( + client, + conn, + graph: Graph, + scenario: Scenario, + metrics: list[str], + *, + use_materialized: bool = True, + parenthesize_filters: bool = True, +) -> tuple[str, dict]: + """DJ's SQL for the scenario's request, and its result keyed by dimensions.""" + filters = [ + render_filter(filter_, lambda column: graph.dimension(column)) + for filter_ in scenario.filters + ] + if parenthesize_filters: + # Combining an unparenthesized OR with another filter is + # known_issues.PRINTER_DROPS_PARENTHESES. + filters = [f"({f})" for f in filters] + response = await client.get( + "/sql/metrics/v3/", + params={ + "metrics": metrics, + "dimensions": [graph.dimension(d) for d in scenario.dimensions], + "filters": filters, + "use_materialized": use_materialized, + }, + ) + assert response.status_code == 200, response.json() + sql = response.json()["sql"] + rows = conn.sql(transpile_to_duckdb(sql)).fetchall() + return sql, by_group(rows, len(scenario.dimensions)) + + +def reference( + conn, + graph: Graph, + scenario: Scenario, + value_expression: str, +) -> tuple[str, dict]: + """A plain join-filter-group query over the raw tables.""" + group_cols = [f"d.{column}" for column in scenario.dimensions] + join = ( + f' {scenario.join_type.upper()} JOIN "default".pbt.{graph.dim_table} d ' + "ON f.k = d.k" + if scenario.dimensions or scenario.filters + else "" + ) + where = ( + " WHERE " + + " AND ".join( + f"({render_filter(filter_, lambda column: f'd.{column}')})" + for filter_ in scenario.filters + ) + if scenario.filters + else "" + ) + group_by = f" GROUP BY {', '.join(group_cols)}" if group_cols else "" + sql = ( + f"SELECT {', '.join([*group_cols, value_expression])} " + f'FROM "default".pbt.{graph.fact_table} f{join}{where}{group_by}' + ) + return sql, by_group(conn.sql(sql).fetchall(), len(group_cols)) + + +def describe(scenario: Scenario, *sql: str) -> str: + return "\n".join( + [ + f"metrics: {[m.definition for m in scenario.metrics]} join: {scenario.join_type}", + f"dimensions: {scenario.dimensions} " + f"filters: {[render_filter(f, str) for f in scenario.filters]}", + *sql, + ], + ) diff --git a/datajunction-server/tests/property/known_issues.py b/datajunction-server/tests/property/known_issues.py new file mode 100644 index 000000000..d7327b514 --- /dev/null +++ b/datajunction-server/tests/property/known_issues.py @@ -0,0 +1,64 @@ +""" +Bugs the property tests have found and that are not yet fixed. + +Generated tests exclude cases affected by known bugs, so they can keep +searching. Each bug also has a small repro marked `xfail(strict=True)`: when +the bug is fixed the repro passes, pytest reports it as XPASS and fails, and +that is the cue to delete the entry here along with its exclusions. +""" + +from dataclasses import dataclass + +import pytest + +from tests.property.comparison import PropertyMismatch + + +@dataclass(frozen=True) +class KnownIssue: + summary: str + issue: str | None = None # link once filed + + @property + def reason(self) -> str: + return self.summary + (f" ({self.issue})" if self.issue else "") + + def xfail(self): + # Only the expected oracle mismatch is an XFAIL. Setup/API assertions + # and unrelated exceptions must still fail the test. + return pytest.mark.xfail( + reason=self.reason, + strict=True, + raises=PropertyMismatch, + ) + + def skip(self): + return pytest.mark.skip(reason=f"known issue: {self.reason}") + + +PRINTER_DROPS_PARENTHESES = KnownIssue( + "SQL assembled from AST nodes built in code prints without the parentheses " + "its nesting needs: sample variance/stddev/correlation combiners, combined " + "filters, derived metrics over compound metrics, and `-(-x)` as `--x`", +) +COVARIANCE_COUNTS_HALF_NULL_ROWS = KnownIssue( + "COVAR_POP/COVAR_SAMP/CORR decompositions count rows where only one of the " + "two arguments is NULL", +) +COUNT_COMPONENT_NAME_COLLISION = KnownIssue( + "COUNT(x) from AVG/COUNT/VAR and from COVAR/CORR share a component name but " + "record the aggregation as `COUNT` vs `COUNT(x)`, so frozen measures conflict", +) +EMPTY_PREAGG_COUNT_IS_NULL = KnownIssue( + "a COUNT served from a pre-aggregation with no matching rows is NULL rather than 0", +) +FILTER_PUSHED_PAST_LEFT_JOIN = KnownIssue( + "a dimension filter is also applied inside the LEFT-joined dimension, so a " + "filter that accepts NULL (e.g. `label IS NULL`) matches facts whose " + "dimension row it filtered out", +) +PREAGG_INNER_JOIN_DROPS_ORPHANS = KnownIssue( + "a pre-aggregation planned with an INNER-linked dimension drops facts with " + "no matching dimension row, so requests that don't use that dimension get " + "a different answer when served from it", +) diff --git a/datajunction-server/tests/property/materialization_test.py b/datajunction-server/tests/property/materialization_test.py new file mode 100644 index 000000000..2b7ac9ced --- /dev/null +++ b/datajunction-server/tests/property/materialization_test.py @@ -0,0 +1,228 @@ +""" +Property: serving a request from a pre-aggregation gives the same answer as +computing it from the raw tables. + +Each example plans a pre-aggregation at a grain covering the request's +dimensions and filters, materializes DJ's planned SQL on DuckDB, reports it +available, then asks for SQL with and without `use_materialized`. + +Both answers come from DJ, so a bug that affects both paths the same way is +invisible here; `random_graph_test` compares against a reference query instead. +""" + +import duckdb +import pytest +from httpx import AsyncClient +from hypothesis import assume, event, given, settings + +from tests.conftest import transpile_to_duckdb +from tests.property.budget import examples +from tests.property.graph import ( + assert_same_results, + build_graph, + create_metric, + describe, + has_orphan_facts, + metrics_sql, + post, +) +from tests.property.known_issues import ( + EMPTY_PREAGG_COUNT_IS_NULL, + PREAGG_INNER_JOIN_DROPS_ORPHANS, +) +from tests.property.scenario import ( + METRIC_BY_DEFINITION, + Scenario, + materialization_scenarios, +) + + +async def check_preaggregated( + client, + conn, + scenario: Scenario, + *, + skip_known: bool, + discard_if_not_served: bool = False, +): + preagg_grain = scenario.preagg_grain + if preagg_grain is None: + raise ValueError("a materialization scenario needs a pre-aggregation grain") + graph = await build_graph(client, conn, scenario) + metrics = [ + await create_metric( + client, + graph, + f"m{i}", + f"SELECT {query.definition} FROM {graph.fact}", + ) + for i, query in enumerate(scenario.metrics) + ] + + plan = await post( + client, + "/preaggs/plan/", + { + "metrics": metrics, + "dimensions": [graph.dimension(d) for d in preagg_grain], + }, + ) + tables = [] + for i, preagg in enumerate(plan["preaggs"]): + table = f"{graph.fact_table}_preagg_{i}" + conn.execute( + f'CREATE TABLE "default".pbt.{table} AS ' + f"{transpile_to_duckdb(preagg['sql'])}", + ) + await post( + client, + f"/preaggs/{preagg['id']}/availability/", + { + "catalog": "default", + "schema_": "pbt", + "table": table, + "valid_through_ts": 20990101, + }, + ) + tables.append(table) + + preagg_sql, preaggregated = await metrics_sql( + client, + conn, + graph, + scenario, + metrics, + ) + raw_sql, raw = await metrics_sql( + client, + conn, + graph, + scenario, + metrics, + use_materialized=False, + ) + served_from_preagg = any(table in preagg_sql for table in tables) + if discard_if_not_served: + # Some otherwise compatible requests still cannot use the planned + # pre-aggregation; retain only examples that exercise substitution. + event(f"served from pre-aggregation: {served_from_preagg}") + assume(served_from_preagg) + else: + assert served_from_preagg, "the planned pre-aggregation was not used" + + if skip_known and () in preaggregated and () in raw: + assume( + not any(p is None and r == 0 for p, r in zip(preaggregated[()], raw[()])), + ) + assert_same_results( + preaggregated, + raw, + describe( + scenario, + f"pre-aggregation grain: {preagg_grain} used: {served_from_preagg}", + "planned SQL:\n" + "\n---\n".join(p["sql"] for p in plan["preaggs"]), + f"SQL using the pre-aggregation:\n{preagg_sql}", + f"SQL from raw tables:\n{raw_sql}", + ), + ) + + +@pytest.mark.asyncio +@settings(max_examples=examples(15)) +@given(scenario=materialization_scenarios()) +async def test_preaggregated_matches_raw( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, + scenario: Scenario, +): + assume( + not ( + scenario.join_type == "inner" + and scenario.preagg_grain + and not scenario.dimensions + and not scenario.filters + and has_orphan_facts(scenario) + ), + ) # known_issues.PREAGG_INNER_JOIN_DROPS_ORPHANS + await check_preaggregated( + module__client_with_roads, + duckdb_conn, + scenario, + skip_known=True, + discard_if_not_served=True, + ) + + +@pytest.mark.asyncio +async def test_eligible_preaggregation_is_used_and_matches_raw( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, +): + """A compatible rollup must route through the available pre-aggregation.""" + await check_preaggregated( + module__client_with_roads, + duckdb_conn, + Scenario( + dimension_rows=((0, "a", 1), (1, "b", 2)), + fact_rows=((0, 1, 2), (0, 2, 3), (1, 3, 4)), + metrics=( + METRIC_BY_DEFINITION["SUM(x)"], + METRIC_BY_DEFINITION["COUNT(x)"], + ), + op="+", + join_type="left", + dimensions=(), + filters=(), + preagg_grain=("label",), + ), + skip_known=False, + ) + + +@pytest.mark.asyncio +@EMPTY_PREAGG_COUNT_IS_NULL.xfail() +async def test_known_issue_count_from_empty_preaggregation( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, +): + await check_preaggregated( + module__client_with_roads, + duckdb_conn, + Scenario( + dimension_rows=((0, None, None),), + fact_rows=(), + metrics=( + METRIC_BY_DEFINITION["SUM(x)"], + METRIC_BY_DEFINITION["COUNT(x)"], + ), + op="+", + join_type="left", + dimensions=(), + filters=(), + preagg_grain=("label",), + ), + skip_known=False, + ) + + +@pytest.mark.asyncio +@PREAGG_INNER_JOIN_DROPS_ORPHANS.xfail() +async def test_known_issue_inner_linked_preaggregation_drops_orphan_facts( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, +): + await check_preaggregated( + module__client_with_roads, + duckdb_conn, + Scenario( + dimension_rows=((0, "a", 1),), + fact_rows=((0, 1, 0), (None, 5, 0)), + metrics=(METRIC_BY_DEFINITION["SUM(x)"],) * 2, + op="+", + join_type="inner", + dimensions=(), + filters=(), + preagg_grain=("label",), + ), + skip_known=False, + ) diff --git a/datajunction-server/tests/property/metric_decomposition_test.py b/datajunction-server/tests/property/metric_decomposition_test.py new file mode 100644 index 000000000..e7032b90a --- /dev/null +++ b/datajunction-server/tests/property/metric_decomposition_test.py @@ -0,0 +1,180 @@ +""" +Property: a decomposed metric, rolled up from a finer grain, equals the metric +computed directly. + +DJ's own decomposition (`MetricComponentExtractor._extract_base`) and component +rendering (`build_component_expression`) produce the SQL; DuckDB executes both +that SQL and the plain aggregate, and the two answers must agree. +""" + +import duckdb +import pytest +import sqlglot +from hypothesis import assume, given, settings +from hypothesis import strategies as st + +from datajunction_server.construction.build_v3.decomposition import ( + build_component_expression, +) +from datajunction_server.sql.decompose import MetricComponentExtractor +from datajunction_server.sql.parsing.backends.antlr4 import parse +from tests.property.budget import examples +from tests.property.comparison import PropertyMismatch, same_value +from tests.property.known_issues import ( + COUNT_COMPONENT_NAME_COLLISION, + COVARIANCE_COUNTS_HALF_NULL_ROWS, + PRINTER_DROPS_PARENTHESES, +) + +METRICS = [ + "SUM(x)", + "MIN(x)", + "MAX(x)", + "COUNT(x)", + "COUNT(*)", + "COUNT_IF(x > 0)", + "AVG(x)", + "VAR_POP(x)", + "VAR_SAMP(x)", + "VARIANCE(x)", + "STDDEV_POP(x)", + "STDDEV_SAMP(x)", + "STDDEV(x)", + "COVAR_POP(x, y)", + "COVAR_SAMP(x, y)", + "CORR(x, y)", +] +BROKEN = { + "VAR_SAMP(x)": PRINTER_DROPS_PARENTHESES, + "VARIANCE(x)": PRINTER_DROPS_PARENTHESES, + "STDDEV_SAMP(x)": PRINTER_DROPS_PARENTHESES, + "STDDEV(x)": PRINTER_DROPS_PARENTHESES, + "COVAR_POP(x, y)": COVARIANCE_COUNTS_HALF_NULL_ROWS, + "COVAR_SAMP(x, y)": COVARIANCE_COUNTS_HALF_NULL_ROWS, + "CORR(x, y)": COVARIANCE_COUNTS_HALF_NULL_ROWS, +} + +values = st.one_of(st.none(), st.integers(-100, 100)) +rows_strategy = st.lists( + st.tuples(st.integers(0, 2), st.integers(0, 3), values, values), + min_size=1, + max_size=40, +) + + +def to_duckdb(spark_sql: str) -> str: + return sqlglot.transpile(spark_sql, read="spark", write="duckdb")[0] + + +def decomposed_sql(metric: str) -> tuple[str, str]: + """DJ's measures query (grain g, p) and its combiner rolled up to grain g.""" + components, derived = MetricComponentExtractor(0)._extract_base( + parse(f"SELECT {metric} FROM t"), + ) + measures = ", ".join( + f"{build_component_expression(c)} AS {c.name}" for c in components + ) + measures_sql = f"SELECT g, p, {measures} FROM t GROUP BY g, p" + combiner = str(derived.select.projection[0]) + return measures_sql, f"SELECT g, {combiner} AS v FROM m GROUP BY g" + + +@pytest.fixture(scope="module") +def conn(): + with duckdb.connect(":memory:") as connection: + yield connection + + +def check_rollup(conn, metric, column_type, rows): + conn.execute( + f"CREATE OR REPLACE TABLE t (g INTEGER, p INTEGER, " + f"x {column_type}, y {column_type})", + ) + conn.executemany("INSERT INTO t VALUES (?, ?, ?, ?)", rows) + + measures_sql, combine_sql = decomposed_sql(metric) + conn.execute(f"CREATE OR REPLACE TABLE m AS {to_duckdb(measures_sql)}") + rolled_up = dict(conn.execute(to_duckdb(combine_sql)).fetchall()) + direct = dict( + conn.execute(f"SELECT g, {metric} FROM t GROUP BY g").fetchall(), + ) + + assert rolled_up.keys() == direct.keys() + mismatches = { + g: (rolled_up[g], direct[g]) + for g in direct + if not same_value(rolled_up[g], direct[g]) + } + if mismatches: + raise PropertyMismatch( + f"{metric}: (rolled-up, direct) by group = {mismatches}\n" + f"measures: {measures_sql}\ncombine: {combine_sql}", + ) + + +@pytest.mark.parametrize("column_type", ["INTEGER", "DOUBLE"]) +@pytest.mark.parametrize( + "metric", + [pytest.param(m, marks=BROKEN[m].skip()) if m in BROKEN else m for m in METRICS], +) +@settings(max_examples=examples(100)) +@given(rows=rows_strategy) +def test_rollup_matches_direct_aggregate(conn, metric, column_type, rows): + check_rollup(conn, metric, column_type, rows) + + +@PRINTER_DROPS_PARENTHESES.xfail() +def test_known_issue_sample_variance_rollup(conn): + check_rollup(conn, "VAR_SAMP(x)", "INTEGER", [(0, 0, 0, None), (0, 0, 1, None)]) + + +@COVARIANCE_COUNTS_HALF_NULL_ROWS.xfail() +def test_known_issue_covariance_with_one_side_null(conn): + check_rollup(conn, "COVAR_POP(x, y)", "INTEGER", [(0, 0, 1, None), (0, 0, 0, 1)]) + + +def components_of(metric: str): + components, _ = MetricComponentExtractor(0)._extract_base( + parse(f"SELECT {metric} FROM t"), + ) + return components + + +def check_component_names(first: str, second: str, *, skip_known: bool) -> None: + by_name = {c.name: c for c in components_of(first)} + for component in components_of(second): + other = by_name.get(component.name) + if other is None: + continue + if skip_known: + spelled_differently = {other.aggregation, component.aggregation} == { + "COUNT", + f"COUNT({component.expression})", + } + assume(not spelled_differently) + if (other.expression, other.aggregation, other.merge) != ( + component.expression, + component.aggregation, + component.merge, + ): + raise PropertyMismatch( + f"`{component.name}`: from {first} it is aggregation=" + f"{other.aggregation!r} over {other.expression!r}; from {second} it is " + f"aggregation={component.aggregation!r} over {component.expression!r}", + ) + + +@settings(max_examples=examples(200)) +@given(first=st.sampled_from(METRICS), second=st.sampled_from(METRICS)) +def test_same_component_name_means_same_component(first, second): + """ + Frozen measures are keyed by component name, so two metrics on the same + parent that produce a component with the same name must define it the same + way, or registering the second metric's measures fails. + """ + check_component_names(first, second, skip_known=True) + + +@COUNT_COMPONENT_NAME_COLLISION.xfail() +def test_known_issue_count_component_collision(): + check_component_names("COUNT(x)", "COVAR_POP(x, y)", skip_known=False) diff --git a/datajunction-server/tests/property/random_graph_test.py b/datajunction-server/tests/property/random_graph_test.py new file mode 100644 index 000000000..d1bbff5d2 --- /dev/null +++ b/datajunction-server/tests/property/random_graph_test.py @@ -0,0 +1,221 @@ +""" +Property: for a randomly generated fact + dimension graph, the SQL DJ builds for +a metric request returns the same rows as a plain hand-written query over the +raw tables. +""" + +from dataclasses import replace + +import duckdb +import pytest +from httpx import AsyncClient +from hypothesis import assume, given, settings + +from tests.property.budget import examples +from tests.property.graph import ( + accepts_missing_dimension, + assert_same_results, + build_graph, + create_metric, + describe, + metrics_sql, + reference, +) +from tests.property.known_issues import ( + FILTER_PUSHED_PAST_LEFT_JOIN, + PRINTER_DROPS_PARENTHESES, +) +from tests.property.scenario import ( + METRIC_BY_DEFINITION, + SIMPLE_METRICS, + FilterAtom, + OrFilter, + Scenario, + scenarios, +) + + +def skip_known_filter_pushdown(conn, scenario: Scenario): + """known_issues.FILTER_PUSHED_PAST_LEFT_JOIN""" + assume( + not ( + scenario.join_type == "left" + and scenario.filters + and accepts_missing_dimension(conn, scenario.filters) + ), + ) + + +async def check_base_metric(client, conn, scenario, *, parenthesize_filters=True): + graph = await build_graph(client, conn, scenario) + query = scenario.metrics[0] + metric = await create_metric( + client, + graph, + "metric", + f"SELECT {query.definition} FROM {graph.fact}", + ) + + dj_sql, dj = await metrics_sql( + client, + conn, + graph, + scenario, + [metric], + parenthesize_filters=parenthesize_filters, + ) + reference_sql, expected = reference(conn, graph, scenario, query.reference) + assert_same_results(dj, expected, describe(scenario, reference_sql, dj_sql)) + + +async def check_derived_metric(client, conn, scenario): + graph = await build_graph(client, conn, scenario) + first, second = scenario.metrics + op = scenario.op + m1 = await create_metric( + client, + graph, + "m1", + f"SELECT {first.definition} FROM {graph.fact}", + ) + m2 = await create_metric( + client, + graph, + "m2", + f"SELECT {second.definition} FROM {graph.fact}", + ) + derived = await create_metric(client, graph, "derived", f"SELECT {m1} {op} {m2}") + + dj_sql, dj = await metrics_sql(client, conn, graph, scenario, [derived]) + # DJ defines metric division as NULL on a zero denominator. Keep that + # contract in the separately written reference query, and compare NULL + # distinctly from infinity or NaN in assert_same_results. + right = f"NULLIF({second.reference}, 0)" if op == "/" else second.reference + reference_sql, expected = reference( + conn, + graph, + scenario, + f"({first.reference}) {op} ({right})", + ) + assert_same_results( + dj, + expected, + describe(scenario, f"derived: m1 {op} m2", reference_sql, dj_sql), + ) + return dj + + +@pytest.mark.asyncio +@settings(max_examples=examples(20)) +@given(scenario=scenarios()) +async def test_base_metric_matches_reference_query( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, + scenario, +): + skip_known_filter_pushdown(duckdb_conn, scenario) + await check_base_metric(module__client_with_roads, duckdb_conn, scenario) + + +@pytest.mark.asyncio +@settings(max_examples=examples(20)) +@given(scenario=scenarios(metric_pool=SIMPLE_METRICS)) +async def test_derived_metric_matches_reference_query( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, + scenario, +): + skip_known_filter_pushdown(duckdb_conn, scenario) + await check_derived_metric(module__client_with_roads, duckdb_conn, scenario) + + +def fixed(**overrides) -> Scenario: + scenario = Scenario( + dimension_rows=((0, "a", 0), (1, "b", 1)), + fact_rows=((0, 1, 0), (1, 2, 0)), + metrics=(METRIC_BY_DEFINITION["SUM(x)"],) * 2, + op="+", + join_type="left", + dimensions=("label",), + filters=(), + ) + return replace(scenario, **overrides) + + +@pytest.mark.asyncio +async def test_derived_metric_division_by_zero_returns_null( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, +): + result = await check_derived_metric( + module__client_with_roads, + duckdb_conn, + fixed( + fact_rows=((0, 1, 0),), + metrics=( + METRIC_BY_DEFINITION["COUNT(*)"], + METRIC_BY_DEFINITION["STDDEV_POP(x)"], + ), + op="/", + dimensions=(), + ), + ) + assert result[()] == (None,) + + +@pytest.mark.asyncio +@PRINTER_DROPS_PARENTHESES.xfail() +async def test_known_issue_or_filter_combined_with_another_filter( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, +): + await check_base_metric( + module__client_with_roads, + duckdb_conn, + fixed( + filters=( + OrFilter(FilterAtom("label", "=", "a"), FilterAtom("label", "=", "b")), + FilterAtom("label", "=", "b"), + ), + ), + parenthesize_filters=False, + ) + + +@pytest.mark.asyncio +@PRINTER_DROPS_PARENTHESES.xfail() +async def test_known_issue_derived_metric_over_compound_metric( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, +): + await check_derived_metric( + module__client_with_roads, + duckdb_conn, + fixed( + fact_rows=((None, -4, 0),), + metrics=( + METRIC_BY_DEFINITION["MAX(x) - MIN(y)"], + METRIC_BY_DEFINITION["SUM(x)"], + ), + op="/", + dimensions=(), + ), + ) + + +@pytest.mark.asyncio +@FILTER_PUSHED_PAST_LEFT_JOIN.xfail() +async def test_known_issue_null_accepting_filter_on_left_joined_dimension( + module__client_with_roads: AsyncClient, + duckdb_conn: duckdb.DuckDBPyConnection, +): + await check_base_metric( + module__client_with_roads, + duckdb_conn, + fixed( + dimension_rows=((0, "b", None), (1, None, None)), + fact_rows=((0, 3, None), (1, 5, None)), + dimensions=("label",), + filters=(FilterAtom("label", "IS NULL"),), + ), + ) diff --git a/datajunction-server/tests/property/scenario.py b/datajunction-server/tests/property/scenario.py new file mode 100644 index 000000000..0f72c18e3 --- /dev/null +++ b/datajunction-server/tests/property/scenario.py @@ -0,0 +1,203 @@ +"""Small, immutable cases shared by the graph and materialization properties.""" + +from __future__ import annotations + +from dataclasses import dataclass, replace +from typing import Literal, TypeAlias +from collections.abc import Callable + +from hypothesis import strategies as st + +DimensionColumn: TypeAlias = Literal["label", "bucket"] +DimensionRow: TypeAlias = tuple[int, str | None, int | None] +FactRow: TypeAlias = tuple[int | None, int | None, int | None] +FilterValue: TypeAlias = str | int + + +@dataclass(frozen=True) +class MetricSpec: + """DJ definition and separately written raw-table reference expression.""" + + definition: str + reference: str + compound_combiner: bool = False + + +# The reference expressions are explicit. Deriving them by editing DJ SQL would +# let a qualification error in the test's oracle hide a DJ error. +# Sample variance/stddev and covariance/correlation remain in fixed known-bug +# repros until their decomposition and printing issues are resolved. +METRICS = ( + MetricSpec("SUM(x)", "SUM(f.x)"), + MetricSpec("COUNT(x)", "COUNT(f.x)"), + MetricSpec("COUNT(*)", "COUNT(*)"), + MetricSpec("MIN(x)", "MIN(f.x)"), + MetricSpec("MAX(x)", "MAX(f.x)"), + MetricSpec("AVG(x)", "AVG(f.x)", compound_combiner=True), + MetricSpec("VAR_POP(x)", "VAR_POP(f.x)", compound_combiner=True), + MetricSpec("STDDEV_POP(x)", "STDDEV_POP(f.x)"), + MetricSpec("COUNT(DISTINCT x)", "COUNT(DISTINCT f.x)"), + MetricSpec("SUM(x + y)", "SUM(f.x + f.y)"), + MetricSpec( + "SUM(CASE WHEN x > 0 THEN x END)", + "SUM(CASE WHEN f.x > 0 THEN f.x END)", + ), + MetricSpec("AVG(x) * 2", "AVG(f.x) * 2", compound_combiner=True), + MetricSpec( + "SUM(x) / COUNT(*)", + "SUM(f.x) / COUNT(*)", + compound_combiner=True, + ), + MetricSpec( + "MAX(x) - MIN(y)", + "MAX(f.x) - MIN(f.y)", + compound_combiner=True, + ), +) +METRIC_BY_DEFINITION = {metric.definition: metric for metric in METRICS} +# Derived metrics currently lose parentheses around compound combiners. +SIMPLE_METRICS = tuple(metric for metric in METRICS if not metric.compound_combiner) + + +@dataclass(frozen=True) +class FilterAtom: + column: DimensionColumn + operator: Literal["=", "<>", ">", "IS NULL", "IS NOT NULL", "IN"] + value: FilterValue | tuple[FilterValue, ...] | None = None + + +@dataclass(frozen=True) +class OrFilter: + left: FilterSpec + right: FilterSpec + + +@dataclass(frozen=True) +class NotFilter: + inner: FilterSpec + + +FilterSpec: TypeAlias = FilterAtom | OrFilter | NotFilter + + +def _literal(value: FilterValue) -> str: + if isinstance(value, str): + return "'" + value.replace("'", "''") + "'" + return str(value) + + +def render_filter( + filter_: FilterSpec, + column_ref: Callable[[DimensionColumn], str], +) -> str: + """Render a predicate with DJ names or physical table aliases.""" + if isinstance(filter_, OrFilter): + return ( + f"{render_filter(filter_.left, column_ref)} OR " + f"{render_filter(filter_.right, column_ref)}" + ) + if isinstance(filter_, NotFilter): + return f"NOT ({render_filter(filter_.inner, column_ref)})" + + column = column_ref(filter_.column) + if filter_.operator in ("IS NULL", "IS NOT NULL"): + return f"{column} {filter_.operator}" + if filter_.operator == "IN": + assert isinstance(filter_.value, tuple) + values = ", ".join(_literal(value) for value in filter_.value) + return f"{column} IN ({values})" + assert isinstance(filter_.value, (str, int)) + return f"{column} {filter_.operator} {_literal(filter_.value)}" + + +def filter_columns(filter_: FilterSpec) -> set[DimensionColumn]: + """Dimension columns a predicate needs to remain available after rollup.""" + if isinstance(filter_, OrFilter): + return filter_columns(filter_.left) | filter_columns(filter_.right) + if isinstance(filter_, NotFilter): + return filter_columns(filter_.inner) + return {filter_.column} + + +@dataclass(frozen=True) +class Scenario: + dimension_rows: tuple[DimensionRow, ...] + fact_rows: tuple[FactRow, ...] + metrics: tuple[MetricSpec, MetricSpec] + op: Literal["+", "-", "*", "/"] + join_type: Literal["left", "inner"] + dimensions: tuple[DimensionColumn, ...] + filters: tuple[FilterSpec, ...] + preagg_grain: tuple[DimensionColumn, ...] | None = None + + +DIMENSION_COLUMNS: tuple[DimensionColumn, ...] = ("label", "bucket") +ATOMS = ( + FilterAtom("label", "=", "a"), + FilterAtom("label", "<>", "b"), + FilterAtom("label", "IS NULL"), + FilterAtom("label", "IS NOT NULL"), + FilterAtom("label", "IN", ("a", "c")), + FilterAtom("bucket", "=", 1), + FilterAtom("bucket", ">", 0), + FilterAtom("bucket", "IS NULL"), + FilterAtom("bucket", "IN", (0, 2)), +) + +labels = st.one_of(st.none(), st.sampled_from(["a", "b", "c"])) +buckets = st.one_of(st.none(), st.integers(0, 2)) +values = st.one_of(st.none(), st.integers(-5, 5)) +atom = st.sampled_from(ATOMS) +filters = st.one_of( + atom, + st.tuples(atom, atom).map(lambda pair: OrFilter(*pair)), + atom.map(NotFilter), +) +dimension_subsets = st.lists( + st.sampled_from(DIMENSION_COLUMNS), + unique=True, + max_size=len(DIMENSION_COLUMNS), +).map(tuple) + + +@st.composite +def scenarios(draw, metric_pool=METRICS): + n_dims = draw(st.integers(1, 3)) + return Scenario( + dimension_rows=tuple( + (key, draw(labels), draw(buckets)) for key in range(n_dims) + ), + fact_rows=tuple( + draw( + st.lists( + st.tuples( + st.one_of(st.none(), st.integers(0, n_dims)), + values, + values, + ), + max_size=10, + ), + ), + ), + metrics=draw( + st.tuples(st.sampled_from(metric_pool), st.sampled_from(metric_pool)), + ), + op=draw(st.sampled_from(["+", "-", "*", "/"])), + join_type=draw(st.sampled_from(["left", "inner"])), + dimensions=draw(dimension_subsets), + filters=tuple(draw(st.lists(filters, max_size=2))), + ) + + +@st.composite +def materialization_scenarios(draw): + scenario = draw(scenarios()) + required = set(scenario.dimensions) + for filter_ in scenario.filters: + required.update(filter_columns(filter_)) + grain = tuple( + column + for column in DIMENSION_COLUMNS + if column in required or draw(st.booleans()) + ) + return replace(scenario, preagg_grain=grain) diff --git a/datajunction-server/tests/property/sql_roundtrip_test.py b/datajunction-server/tests/property/sql_roundtrip_test.py new file mode 100644 index 000000000..4d3f4685e --- /dev/null +++ b/datajunction-server/tests/property/sql_roundtrip_test.py @@ -0,0 +1,188 @@ +""" +Property: DJ's parser and printer preserve what an expression means. + +Random expressions are parsed and printed by DJ, then the original and DJ's +version are both evaluated on DuckDB over random rows. Any difference in the +results means DJ changed the query's meaning. +""" + +import re + +import duckdb +import pytest +from hypothesis import assume, given, settings +from hypothesis import strategies as st + +from datajunction_server.sql.parsing.backends.antlr4 import parse +from tests.property.budget import examples + +COLUMNS = ["a", "b", "c"] + +numeric_leaf = st.one_of( + st.sampled_from(COLUMNS), + st.integers(-5, 5).map(str), + st.just("NULL"), +) + + +def numeric_branches(children): + return st.one_of( + st.tuples(children, st.sampled_from(["+", "-", "*"]), children).map( + lambda t: f"{t[0]} {t[1]} {t[2]}", + ), + children.map(lambda e: f"- {e}" if e.startswith("-") else f"-{e}"), + children.map(lambda e: f"- {e}"), + children.map(lambda e: f"({e})"), + st.tuples(children, children).map(lambda t: f"COALESCE({t[0]}, {t[1]})"), + children.map(lambda e: f"ABS({e})"), + st.tuples(boolean_expr(children), children, children).map( + lambda t: f"CASE WHEN {t[0]} THEN {t[1]} ELSE {t[2]} END", + ), + children.map(lambda e: f"CAST({e} AS BIGINT)"), + ) + + +def boolean_expr(numeric): + comparison = st.tuples( + numeric, + st.sampled_from(["=", "<>", "<", "<=", ">", ">="]), + numeric, + ).map(lambda t: f"{t[0]} {t[1]} {t[2]}") + atoms = st.one_of( + comparison, + numeric.map(lambda e: f"{e} IS NULL"), + numeric.map(lambda e: f"{e} IS NOT NULL"), + st.tuples(numeric, numeric, numeric).map( + lambda t: f"{t[0]} BETWEEN {t[1]} AND {t[2]}", + ), + st.tuples(numeric, st.lists(numeric, min_size=1, max_size=3)).map( + lambda t: f"{t[0]} IN ({', '.join(t[1])})", + ), + ) + return st.recursive( + atoms, + lambda inner: st.one_of( + st.tuples(inner, st.sampled_from(["AND", "OR"]), inner).map( + lambda t: f"{t[0]} {t[1]} {t[2]}", + ), + inner.map(lambda e: f"NOT {e}"), + inner.map(lambda e: f"({e})"), + ), + max_leaves=4, + ) + + +numeric_expr = st.recursive(numeric_leaf, numeric_branches, max_leaves=6) + +string_leaf = st.one_of( + st.just("s"), + st.sampled_from(["''", "'x'", "'it''s'", "'a%'", "'%_'", "'\\\\'"]), +) +string_expr = st.recursive( + string_leaf, + lambda inner: st.one_of( + st.tuples(inner, inner).map(lambda t: f"CONCAT({t[0]}, {t[1]})"), + inner.map(lambda e: f"UPPER({e})"), + st.tuples(boolean_expr(numeric_leaf), inner, inner).map( + lambda t: f"CASE WHEN {t[0]} THEN {t[1]} ELSE {t[2]} END", + ), + ), + max_leaves=4, +) +string_predicates = st.one_of( + st.tuples(string_expr, string_expr).map(lambda t: f"{t[0]} = {t[1]}"), + st.tuples(string_expr, string_expr).map(lambda t: f"{t[0]} LIKE {t[1]}"), + st.tuples(string_expr, string_expr).map(lambda t: f"{t[0]} NOT LIKE {t[1]}"), +) +multi_case = st.tuples( + st.lists( + st.tuples(boolean_expr(numeric_leaf), numeric_leaf), + min_size=1, + max_size=3, + ), + st.one_of(st.none(), numeric_leaf), +).map( + lambda t: ( + "CASE " + + " ".join(f"WHEN {w} THEN {v}" for w, v in t[0]) + + (f" ELSE {t[1]}" if t[1] is not None else "") + + " END" + ), +) + +expressions = st.one_of( + numeric_expr, + boolean_expr(numeric_expr), + string_expr, + string_predicates, + multi_case, +) + +# DJ prints `-(-x)` as `--x`, which SQL reads as a comment +# (known_issues.PRINTER_DROPS_PARENTHESES); skipped so it doesn't mask other +# differences. +KNOWN_DOUBLE_MINUS = "--" + +rows = st.lists( + st.tuples( + *[st.one_of(st.none(), st.integers(-3, 3))] * len(COLUMNS), + st.one_of(st.none(), st.sampled_from(["", "x", "it's", "a%", "\\", "X"])), + ), + min_size=1, + max_size=6, +) + + +def evaluate(conn, expression: str): + return conn.execute(f"SELECT {expression} FROM t ORDER BY rowid").fetchall() + + +@pytest.fixture(scope="module") +def conn(): + with duckdb.connect(":memory:") as connection: + yield connection + + +def dj_roundtrip(expression: str) -> str: + query = parse(f"SELECT {expression} FROM t") + return str(query.select.projection[0]) + + +@settings(max_examples=examples(1000)) +@given(expression=expressions, data=rows) +def test_dj_printing_preserves_meaning(conn, expression, data): + conn.execute( + "CREATE OR REPLACE TABLE t (a INTEGER, b INTEGER, c INTEGER, s VARCHAR)", + ) + conn.executemany("INSERT INTO t VALUES (?, ?, ?, ?)", data) + try: + expected = evaluate(conn, expression) + except duckdb.Error: + assume(False) + + printed = dj_roundtrip(expression) + assume(KNOWN_DOUBLE_MINUS not in printed) + try: + actual = evaluate(conn, printed) + except duckdb.Error as exc: + raise AssertionError( + f"DJ's version no longer runs.\n original: {expression}\n" + f" DJ: {printed}\n error: {exc}", + ) from exc + assert actual == expected, ( + f"\n original: {expression}\n DJ: {printed}\n" + f" expected: {expected}\n got: {actual}" + ) + + +def without_negative_zero(sql: str) -> str: + # DJ reads the literal `-0` as 0; the value is the same either way. + return re.sub(r"(?=3.11, <3.14" resolution-markers = [ "python_full_version >= '3.13'", @@ -926,6 +926,7 @@ test = [ { name = "gevent" }, { name = "greenlet" }, { name = "httpx" }, + { name = "hypothesis" }, { name = "mcp" }, { name = "pre-commit" }, { name = "pytest" }, @@ -1001,6 +1002,7 @@ test = [ { name = "gevent", specifier = ">=24.2.1" }, { name = "greenlet", specifier = ">=3.0.3" }, { name = "httpx", specifier = ">=0.27.0" }, + { name = "hypothesis", specifier = ">=6.168.1" }, { name = "mcp", specifier = ">=1.0.0" }, { name = "pre-commit", specifier = ">=3.2.2" }, { name = "pytest", specifier = ">=7.3.0" }, @@ -1571,6 +1573,79 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/d2/fd/6668e5aec43ab844de6fc74927e155a3b37bf40d7c3790e49fc0406b6578/httpx_sse-0.4.3-py3-none-any.whl", hash = "sha256:0ac1c9fe3c0afad2e0ebb25a934a59f4c7823b60792691f779fad2c5568830fc", size = 8960, upload-time = "2025-10-10T21:48:21.158Z" }, ] +[[package]] +name = "hypothesis" +version = "6.168.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "sortedcontainers" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/aa/31/b54cd138ee36a32d0047d40fe7bdcb63cefd57ce098a26024161332d15fd/hypothesis-6.168.1.tar.gz", hash = "sha256:fd8acb5c67f260dfbdcc1b2dc95ff48b2b3f90708ae595c76f591bbf3f6f8eeb", size = 510823, upload-time = "2026-09-23T04:36:04.09Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/48/95/3091b9ad24baa9666d49feea433a8c4060ad47245f0d529d736af3672ddd/hypothesis-6.168.1-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:5d9cc2d2a4779c92788389564fc6760df7c2909185f6431378133dea6933194f", size = 791345, upload-time = "2026-09-23T04:35:39.379Z" }, + { url = "https://files.pythonhosted.org/packages/c5/b3/b954ffa69f91d111f801030c28d6721f422222af245a846571fc87874242/hypothesis-6.168.1-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:d0e1ec15d32af11da6784d2506de0c901e4c1abf97a3cdcfc144225457783f4e", size = 787104, upload-time = "2026-09-23T04:35:21.799Z" }, + { url = "https://files.pythonhosted.org/packages/e5/dc/732df08f570d845a480d3729bfb46beccc441bfabe2804bb3a458649be73/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4cc38519a57c4b919b2c59714a67501d20b46b0c96a0078cbdc9f86e728efce3", size = 1123848, upload-time = "2026-09-23T04:33:48.857Z" }, + { url = "https://files.pythonhosted.org/packages/88/06/e1e937df42243267ba25ede5e9b29921186f290f6c0515729481cc41726e/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:7e229e937e761830c7fa6437fb12cb801616a1737af0898abf5edba57d2ad691", size = 1147698, upload-time = "2026-09-23T04:35:33.439Z" }, + { url = "https://files.pythonhosted.org/packages/be/d1/3b41c008dc2a49e43bc4893dcd27ea9d9fef64fef50382ca5f7107680b1f/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:569ba21e5543840895d17633cc6716a9578ac83b1193a2978b3f24cc16b93085", size = 1149280, upload-time = "2026-09-23T04:35:45.514Z" }, + { url = "https://files.pythonhosted.org/packages/71/ea/d54781058fe6a487b81ec6c636f3806fff6a417b48cf3fcd2b003d80f3ce/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:0a232665716607c3c50c9a3aeeac5929be1d4e58703f5e7cb51db7099445a8c6", size = 1191693, upload-time = "2026-09-23T04:35:57.896Z" }, + { url = "https://files.pythonhosted.org/packages/4f/f2/43bdd07450978611d25989ffdcbe1579acd46c1500a484f8838359fe7880/hypothesis-6.168.1-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9022fe330d8839259da9b496573c4b83ea248bf32029eba23ff589f3c25754bf", size = 1169725, upload-time = "2026-09-23T04:34:32.59Z" }, + { url = "https://files.pythonhosted.org/packages/d1/72/fd4b657cba66940f208bc0eccaceaa777553009d26e86068ff3ab5e55b0b/hypothesis-6.168.1-cp310-abi3-manylinux_2_31_riscv64.whl", hash = "sha256:69be660b831d55ab35619b30abd311a6cb8cd0068a1192210d194a776b211094", size = 1129181, upload-time = "2026-09-23T04:35:27.495Z" }, + { url = "https://files.pythonhosted.org/packages/d6/3f/516a29d7a2bd31e28972de635350eab086c94faa07afcdd3ad0f06e70db8/hypothesis-6.168.1-cp310-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:42e2cccca80ee4ffe8f66edf5b2adb2a16b8cd3c5110b3a774d70f2fc75a42aa", size = 1160162, upload-time = "2026-09-23T04:35:29.333Z" }, + { url = "https://files.pythonhosted.org/packages/54/09/e2c7e32f281b31d4c19cc2fa935bfdde5b31ad4848a485e5ebb8de2389da/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:11cbd2a2539194f0960de4ed9f980882e3a3f55c476afcc6f4cc342eec5a07d2", size = 1299714, upload-time = "2026-09-23T04:35:12.529Z" }, + { url = "https://files.pythonhosted.org/packages/ba/bc/85850bdb1fc35d0fa9e299ca3b2e354f403959666c7ef4d0bfda9c308eee/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:aa1620111616c660a1db03347b1790b5914ef30f86c6e8ddd6c765785291f0b4", size = 1425335, upload-time = "2026-09-23T04:35:23.557Z" }, + { url = "https://files.pythonhosted.org/packages/40/c0/66bdb24f32d32aa5636655b6037af07962118b89fae16ee74bc9a1bb982e/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_i686.whl", hash = "sha256:a07c8665c95bb856f99ecd60b6f50d94dcf8157415387c32dcd5bb5b5061d032", size = 1376916, upload-time = "2026-09-23T04:34:59.742Z" }, + { url = "https://files.pythonhosted.org/packages/31/f0/7ad0b17bda81c473cbc539009148f3e0950880c4618028691bcf62eeb8d2/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:8a50b6335d5fa2312dd9f5d50faeb953e61475b301d2c073cd7a4c5efd7e6171", size = 1280975, upload-time = "2026-09-23T04:34:21.966Z" }, + { url = "https://files.pythonhosted.org/packages/e2/01/8d04d1eecf03a6097e26400e8393eff7a81fd52c877a664ee4e97adf7880/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:4ccb0276db8ad197a63dff25c4868e11123346347df6900625d2b1f794ce7c19", size = 1300238, upload-time = "2026-09-23T04:34:52.825Z" }, + { url = "https://files.pythonhosted.org/packages/f5/74/eda0c70b3845789c713fe2ac72d3d4586da83f38a34f1f7aefbda0b5b01e/hypothesis-6.168.1-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:85ca7788574e06d0591d51e19ecbb1d09bc82d50c6eae5de0fe8317590b1e138", size = 1336088, upload-time = "2026-09-23T04:33:42.002Z" }, + { url = "https://files.pythonhosted.org/packages/80/c9/43d8528e6d43eebb7688d36e530d1fc7763809b8f923b62cc8b4d95f99ff/hypothesis-6.168.1-cp310-abi3-win32.whl", hash = "sha256:77378e04aa47a22a615af80f42970dc44782f02b8a82ac105b619a2cc7f5eee2", size = 678004, upload-time = "2026-09-23T04:34:29.564Z" }, + { url = "https://files.pythonhosted.org/packages/8c/8b/f7d4de5b57972f7f68a6c7fb5698f60425589eae6fcb175f285f2525e8c0/hypothesis-6.168.1-cp310-abi3-win_amd64.whl", hash = "sha256:f94d3dbc31112f8fbcb3af19a662ee5e90e380e6ae676357903caa14a1604763", size = 684722, upload-time = "2026-09-23T04:34:41.279Z" }, + { url = "https://files.pythonhosted.org/packages/a5/fd/528eb221a82da0aa477e34c7af6ed2b4a5109b326638d4418b9e49435e96/hypothesis-6.168.1-cp310-abi3-win_arm64.whl", hash = "sha256:9bd7e8c6e2c3f6a5befee6b03ebd2170da2e50658075c6decad184777c72a726", size = 682711, upload-time = "2026-09-23T04:33:46.085Z" }, + { url = "https://files.pythonhosted.org/packages/6d/1b/620484ba32fbb655ccbd02f3c510e7a961e840a466c3672ef0b877ab8716/hypothesis-6.168.1-cp311-cp311-macosx_10_12_x86_64.whl", hash = "sha256:9fd235c4ccfefe18fd97c181bdf247187cb3d4fdc8800a4d623322234ce4cfd3", size = 792049, upload-time = "2026-09-23T04:34:28.21Z" }, + { url = "https://files.pythonhosted.org/packages/03/df/2ad250c2c4ee4d5f43d6d9bc5d852e7b81c771e4c8946212931b8cafe849/hypothesis-6.168.1-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:0722212cf825d001a0d3b4db9303cc31af467f6814b69acab91d91c3f629c95a", size = 787932, upload-time = "2026-09-23T04:35:20.079Z" }, + { url = "https://files.pythonhosted.org/packages/24/69/4f8e8bf1d5ad81352a05aa0835baef82e691e3e43933bafe6d6f2c171763/hypothesis-6.168.1-cp311-cp311-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:94a1053575f133ff6978876fd4b4c2a153c1111f1b5f14d6cfb355c306fc4cb5", size = 1124000, upload-time = "2026-09-23T04:34:14.106Z" }, + { url = "https://files.pythonhosted.org/packages/b6/e7/b72a34e0c6b434d3ad2b2e0040fa88d06a15a7880c59bddacf1e27d4902a/hypothesis-6.168.1-cp311-cp311-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:6d2c74d496304574cfed537e4f657620be7fee3e287381c0aa4b481211e6ec92", size = 1170172, upload-time = "2026-09-23T04:34:30.93Z" }, + { url = "https://files.pythonhosted.org/packages/b6/6c/afe1b73b58479ba0d75e5408259978d580e96fee812eb4ea2bb83a76f7ff/hypothesis-6.168.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:cf4cc6819ba9689effba38dce98a3025554669f76be07e47cc1002e93e8c4918", size = 1300163, upload-time = "2026-09-23T04:35:53.549Z" }, + { url = "https://files.pythonhosted.org/packages/b0/3d/aad54ffe7942c06da79c719fc98daa1dffdd8ac2d384ca64cc3611b1a833/hypothesis-6.168.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:7888e0876e1d93fb92a6a4869834b0d0de18f07f6600edbf3d1cad16cf089357", size = 1336274, upload-time = "2026-09-23T04:35:10.655Z" }, + { url = "https://files.pythonhosted.org/packages/0f/cb/8d2a3474e7921cfb5e59f933ac6fe1ad83b95cc98a871af8f0a68dcf946e/hypothesis-6.168.1-cp311-cp311-win_amd64.whl", hash = "sha256:4aa312b6447743a72283698292ea0e1c4d684279b38dc7f270729c4efda6c27a", size = 684479, upload-time = "2026-09-23T04:33:59.307Z" }, + { url = "https://files.pythonhosted.org/packages/7e/6f/9d5ae55d81e10f0fc46868c19ab9d468e9959e82ff8cc8c26990b16114f3/hypothesis-6.168.1-cp312-cp312-macosx_10_12_x86_64.whl", hash = "sha256:dfe14a88b1ed03ce47333d800cc37949da73feb0cd9767e7ef076d2202768928", size = 793124, upload-time = "2026-09-23T04:33:52.589Z" }, + { url = "https://files.pythonhosted.org/packages/14/11/54c19d259d7f92176aa2199c90f7951e4e396a26898a596dc875ccb9c389/hypothesis-6.168.1-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:8acd000c3bf233cdcd28c1ff03f1bf56cb03821b1c742d00d57978b20fc6e6f8", size = 784674, upload-time = "2026-09-23T04:34:38.137Z" }, + { url = "https://files.pythonhosted.org/packages/5f/ae/f80a9810f88bd654ab0d0ed0652c274e4c0384e22d4f9442b6753b9200a6/hypothesis-6.168.1-cp312-cp312-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:29280d4743546ddcf74644f9acdc5726972cba9182699a86b7e77f3f8f6e5dde", size = 1122860, upload-time = "2026-09-23T04:33:51.375Z" }, + { url = "https://files.pythonhosted.org/packages/f4/5d/aa5b7ffd57ec475a9252c55985c07a366823c3f85267d44648e94c1ec321/hypothesis-6.168.1-cp312-cp312-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:bd9a488d46b8c7da3103058c7569376a3a08994aee9a7d6f1d160d3d3b49b82d", size = 1168922, upload-time = "2026-09-23T04:34:44.567Z" }, + { url = "https://files.pythonhosted.org/packages/65/10/4862448192d4c9cc61a3b8a620730800f56b1e83b0bfda63dbc728526eb3/hypothesis-6.168.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:451ee6c42955887a9e1850c53813e31182eab1663e4d3dacea4e535dedc45b5d", size = 1298568, upload-time = "2026-09-23T04:35:31.191Z" }, + { url = "https://files.pythonhosted.org/packages/15/ad/55e73fda9afbfcf3336b2eee564833bc9a4fee1f7a32fe5f9d183742da39/hypothesis-6.168.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:bc2c036f4a9534af64977e38b2976295136f83417fdb95bed816a52cc3023497", size = 1335091, upload-time = "2026-09-23T04:34:23.42Z" }, + { url = "https://files.pythonhosted.org/packages/ef/99/b6340fa21a89b0088ed023fe984109e1c47a07dfd741fd6d8e17a9e2795f/hypothesis-6.168.1-cp312-cp312-win_amd64.whl", hash = "sha256:d603d7c96d030a63701b6ad9bff0bc4870e4c226335b3db509f1eed1f7fc061b", size = 682011, upload-time = "2026-09-23T04:34:34.565Z" }, + { url = "https://files.pythonhosted.org/packages/f5/23/093b9dc768e89d2cec74d29115f4004fe1be64218bac65f8d817d727d30b/hypothesis-6.168.1-cp313-cp313-macosx_10_12_x86_64.whl", hash = "sha256:aebf7e9d7ba9f920e154abc65851d4affab497daeaea285474bed93840a983ea", size = 793062, upload-time = "2026-09-23T04:33:58.177Z" }, + { url = "https://files.pythonhosted.org/packages/df/d7/f191686b44b3f892ae1404e238a55260152a645a2deafe8b2949c7e90fe9/hypothesis-6.168.1-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:b26616604cac458dd83e75900fefd9190f1074e67d5d9dc4d9e321d30f295eca", size = 784518, upload-time = "2026-09-23T04:34:02.012Z" }, + { url = "https://files.pythonhosted.org/packages/10/25/b7a5dca1911c9012a564d0b7aaeb44e393ade9424a695dea179ff32d79fe/hypothesis-6.168.1-cp313-cp313-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:cb6d89cff2edcd45e5af957c7b6a71ae7887419561f7b9c7055e1c50dea41017", size = 1122844, upload-time = "2026-09-23T04:35:06.897Z" }, + { url = "https://files.pythonhosted.org/packages/ea/60/ddacd0d244e6719ea9a9d322d4d4b6b054fd27ada94ec7aa7dce7ef11297/hypothesis-6.168.1-cp313-cp313-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:2a4ce6310ec21000446ae75cdc6667b81d4ee0f2d226a0dfaf8baca42dd151d0", size = 1168779, upload-time = "2026-09-23T04:33:56.885Z" }, + { url = "https://files.pythonhosted.org/packages/dc/07/343bab6676a41f38cb7c5912ab3ca059258bf40e41c181d1b77cc7251801/hypothesis-6.168.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:a22c6e730c327d22a4d61b53f66639cdf6f0fe4a5a1b3595c36533e5b8127169", size = 1298434, upload-time = "2026-09-23T04:34:48.576Z" }, + { url = "https://files.pythonhosted.org/packages/7d/bd/4b72fafa5f46a455024a9242b62505b1a53fad95ace162e831c4ffb233b9/hypothesis-6.168.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:dcd66493fe180d38641b02dd0456cc1e429655f8940dd1396ab874808f16dd34", size = 1334975, upload-time = "2026-09-23T04:35:01.531Z" }, + { url = "https://files.pythonhosted.org/packages/3f/ae/48643915cda400ca2dccb9bcdc363690b13bcd5c179172bc6450c3d76ce8/hypothesis-6.168.1-cp313-cp313-win_amd64.whl", hash = "sha256:a8a96f2e9fa2a57f95865aaf092943775e7ea96f72535371b2be20fe5464298c", size = 681978, upload-time = "2026-09-23T04:36:00.056Z" }, + { url = "https://files.pythonhosted.org/packages/54/e9/45595951f8bc9bd03f032325e7b176ca990a0c86bf2e09d481e8ee5a5cab/hypothesis-6.168.1-cp315-abi3.abi3t-macosx_10_12_x86_64.whl", hash = "sha256:e9510e33467d586a7700d56b9a6ee65b44a06832c1cab3d270369d7c53381276", size = 791046, upload-time = "2026-09-23T04:35:55.724Z" }, + { url = "https://files.pythonhosted.org/packages/8d/3e/718431d0ab72c4c07454f71040b4de298a63e765f20088557dc1bef0a48b/hypothesis-6.168.1-cp315-abi3.abi3t-macosx_11_0_arm64.whl", hash = "sha256:99970e3a75f11dde0414d3d6724bd8b7a2bd1deeb9a509ec9236c91d41a5cff0", size = 782982, upload-time = "2026-09-23T04:34:06.017Z" }, + { url = "https://files.pythonhosted.org/packages/d7/05/c60db6df1456a70a3c589e42acaa446c22f56ed342c1b8dd04b8e49df8fd/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:a5ea78773de5bcc25266e37eb086472e90a85fc81b8d4a9ca8afc15858ee15fb", size = 1120961, upload-time = "2026-09-23T04:33:43.494Z" }, + { url = "https://files.pythonhosted.org/packages/8b/fe/d943250395750ecc3b4ec95b1ab3c8119f6cbad08c1ba9169ffee7d89049/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:13f502ce2423e46dfcd2af38950a86ecf97084d94e721f7f13dd5a70afb36e0a", size = 1143877, upload-time = "2026-09-23T04:34:56.065Z" }, + { url = "https://files.pythonhosted.org/packages/d3/9d/8c1f8e11f20fd81ce40caed7ad6d6a9776e24775070e87f3aa9ba3683272/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:69b4bcaa8b0e753fc8921011d5d10c8a7c3caf814c6a968bc140a08917973fdc", size = 1146503, upload-time = "2026-09-23T04:34:04.791Z" }, + { url = "https://files.pythonhosted.org/packages/71/d8/ac2e9fa653d2f05a13724837007e857d376d81fdf728543ab85f6c7f7350/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:7ed811908d38ca8e05c20ad25ecb61edc1e20a519667d4b668be49dd09e6bf3e", size = 1189216, upload-time = "2026-09-23T04:34:36.271Z" }, + { url = "https://files.pythonhosted.org/packages/2b/7b/640952448781c7df5098880e2e2d6e7702b3a1c0badd4c0ea5fca6d6ab53/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:9239c868418088d58ae18d041f9f424bd7a75dd143f209c35fde38770b65f74f", size = 1166919, upload-time = "2026-09-23T04:35:16.498Z" }, + { url = "https://files.pythonhosted.org/packages/b6/de/0d46b330a28f7614850fb1a0564cf88af7c0b29cbd791e9d04ade7f66f13/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_31_riscv64.whl", hash = "sha256:5cac279798ecc92d3301ec4421c0458e22627f8a79fdde8a6c98ab1d4b8811f6", size = 1126652, upload-time = "2026-09-23T04:35:49.432Z" }, + { url = "https://files.pythonhosted.org/packages/92/36/6adeca1724ed4966095406ffe7308fcff338cf9f5b7443d5ea34a01ab6e1/hypothesis-6.168.1-cp315-abi3.abi3t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:6bc660dc14d633641e22e407417d35bf2925331273ed71bc6ab5536dfca5c72d", size = 1155687, upload-time = "2026-09-23T04:35:18.234Z" }, + { url = "https://files.pythonhosted.org/packages/bd/00/e9b7c7e4a0a75d1621be95a126999352e01a16d68b0a59b396695c7b4991/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_aarch64.whl", hash = "sha256:157a29e4d6bcb77ae732cb83d6546beef47265ecac88408fdd5d5b2bd325685d", size = 1296471, upload-time = "2026-09-23T04:34:12.763Z" }, + { url = "https://files.pythonhosted.org/packages/8f/d6/52cb320e9e07cc67f718506617b5900344370394e39f4affada317aab313/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_armv7l.whl", hash = "sha256:b125aac89594ee2a76687b724f057b2423ea83dfee6b8ff59f54289d6eef19a1", size = 1421838, upload-time = "2026-09-23T04:34:20.451Z" }, + { url = "https://files.pythonhosted.org/packages/9b/93/969c33a87cd2865beffebe491294b8115e7e7421c6840f79a67486535bfc/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_i686.whl", hash = "sha256:4b960c9d31b85dbc6a25888897c02bbea5f498844932841612925184fc78d9eb", size = 1373994, upload-time = "2026-09-23T04:34:00.608Z" }, + { url = "https://files.pythonhosted.org/packages/34/c9/5fe4ade4bd23baa2e4cd27668bb8ac10e840ce19fbcfc74116ea529cbaa7/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_ppc64le.whl", hash = "sha256:d4d76f35874471fe9b8e1bb696fb532363ef3d9588161a06d30520f638f15a39", size = 1278216, upload-time = "2026-09-23T04:34:07.253Z" }, + { url = "https://files.pythonhosted.org/packages/a1/17/30b9a1ad975e90700dea74ce8b08ab07b9e58c4d83918422bc8841942c1e/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_riscv64.whl", hash = "sha256:d5147c5f7497d109eed0267c498adaf5cad6e4beeaca6fad6057f2007bae70a7", size = 1297647, upload-time = "2026-09-23T04:35:51.491Z" }, + { url = "https://files.pythonhosted.org/packages/e0/2f/233a9d5adafad6e187ab8f4d9275592cf120b56e63ea1cfb02213c01021e/hypothesis-6.168.1-cp315-abi3.abi3t-musllinux_1_2_x86_64.whl", hash = "sha256:a909f89758e184c4f86418598de297a2ced5b922979c0f98c5d73ff6bafe38e5", size = 1333790, upload-time = "2026-09-23T04:34:42.828Z" }, + { url = "https://files.pythonhosted.org/packages/10/d7/4c9d4d597f70316d781cd99b600adae090a6f5d675487196e10510a23afe/hypothesis-6.168.1-cp315-abi3.abi3t-win32.whl", hash = "sha256:66b9fb6f02840f8ffa2a4e6359aff6ecc1b2c6630c8a2066771044fa22956c7a", size = 675206, upload-time = "2026-09-23T04:35:14.592Z" }, + { url = "https://files.pythonhosted.org/packages/b4/a8/f9fc0c0dbb9e80238fc7145cb4ce52e0897ea417a23217022911855bccda/hypothesis-6.168.1-cp315-abi3.abi3t-win_amd64.whl", hash = "sha256:9246eaa80044c33a182be661cbed7e6a7314c61359ef501ea9e9efd90d50ca3b", size = 681503, upload-time = "2026-09-23T04:34:11.223Z" }, + { url = "https://files.pythonhosted.org/packages/11/b4/962ae350c10860be37c9abf98c0ef1ef23e4baca3ee7ad77208941e370e5/hypothesis-6.168.1-cp315-abi3.abi3t-win_arm64.whl", hash = "sha256:52e0c804cbf697d4760400c84444a91cbd5dfe7ae0ef1bb3228f74fb5547eb9e", size = 679199, upload-time = "2026-09-23T04:34:03.472Z" }, + { url = "https://files.pythonhosted.org/packages/aa/c5/2bfdcc601b08eb9611bc524d18ca0a8cdbbdd7bd07bc68c38e0196a00668/hypothesis-6.168.1-pp311-pypy311_pp73-macosx_10_12_x86_64.whl", hash = "sha256:91e3700d9c35e184dd253cff2f151f890557242bede75a2535901e9062d42975", size = 792943, upload-time = "2026-09-23T04:35:25.308Z" }, + { url = "https://files.pythonhosted.org/packages/4b/52/36ae5128d9102d73dc578cd7b241c33f5483bd56e05799a2d06043a1ce7d/hypothesis-6.168.1-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:c19bc5da324a5899527ea749522e9ceb47f1eb54772869673abe59ae0b703aef", size = 788802, upload-time = "2026-09-23T04:34:39.655Z" }, + { url = "https://files.pythonhosted.org/packages/23/34/698fc3481757503dc60f07547c29098dce97ac035a1ee134d66d88814281/hypothesis-6.168.1-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4a4c9e46b2f349658d3013a357e1fbf114bc08abced7ee8e91cd6e0bc718b195", size = 1124766, upload-time = "2026-09-23T04:34:57.869Z" }, + { url = "https://files.pythonhosted.org/packages/02/c9/1f871666106d90048027fc6b309ec505f0fb44e4519add73ddaad98bcd9a/hypothesis-6.168.1-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:95a0beee16fcbc0516951f2ef3ef1a479ad11d9726da49840139303b36ebc1d0", size = 1171766, upload-time = "2026-09-23T04:34:54.476Z" }, + { url = "https://files.pythonhosted.org/packages/17/44/eff662526259ac4447dde4abe4fc285d0e8e5ff5345d35af836c6f0f12cc/hypothesis-6.168.1-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:4a8beb513c066fcb187b885fdfa93da4006f223d50d605b687bfb7e60aac33e9", size = 685449, upload-time = "2026-09-23T04:33:44.837Z" }, +] + [[package]] name = "identify" version = "2.6.16"