diff --git a/AGENTS.md b/AGENTS.md index 51c30048..88b9fde9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -82,7 +82,8 @@ uv run --env-file .env pytest -n 1 tests/pyathena/test_cursor.py -v #### Test Conventions - **Class-based tests** for integration tests that use fixtures (cursors, engines): `class TestCursor:` with methods like `def test_fetchone(self, cursor):` -- **Standalone functions** for unit tests of pure logic (converters, parsers, utils): `def test_to_struct_json_formats(input_value, expected):` +- **Standalone functions** are the default for stateless helpers: `def test_to_struct_json_formats(input_value, expected):`. Unit tests may use a class when it groups the behavior of one object or meaningful common setup, such as `TestTypeSignatureParser`. +- Function-oriented utility tests may remain standalone even when they use AWS fixtures, as in `tests/pyathena/pandas/test_util.py`. Fixture use alone does not determine the grouping. - Test file naming mirrors source: `pyathena/parser.py` → `tests/pyathena/test_parser.py` - **Fixtures**: Cursor/engine fixtures are defined in `conftest.py` and injected by name (e.g., `cursor`, `engine`, `async_cursor`). Use `indirect=True` parametrization to pass connection options: @@ -93,6 +94,9 @@ uv run --env-file .env pytest -n 1 tests/pyathena/test_cursor.py -v ``` - **Parametrize** with `@pytest.mark.parametrize(("input", "expected"), [...])` for data-driven tests +- Keep parameter definitions declarative. Build mocks, configured result sets, and one-shot readers during fixture setup or test execution. Pure values and framework type/expression objects may be constructed in parameter definitions. +- Make expected values independent of the implementation under test. A library comparison is appropriate when matching that library is the contract; describe intentional differences with explicit expected values. +- Preserve test IDs, parameter coverage, marks, fixture scopes, and resource cleanup when reorganizing tests. SQLAlchemy compliance tests retain their upstream class and plugin conventions and applicable attribution. - **Integration tests** (need AWS) use cursor/engine fixtures with real Athena queries; **unit tests** (no AWS) call functions directly with test data ### Markdown Lint diff --git a/docs/testing.md b/docs/testing.md index 575d1676..db559fcc 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -128,6 +128,38 @@ uv run --env-file .env pytest -n 1 tests/pyathena/test_cursor.py -v ``` A targeted run helps during development but does not replace other coverage required by the affected callers or features. + +(testing-offline)= + +### Run self-contained tests offline + +The pandas and Polars result-set modules have self-contained tests that can run without AWS access when the session hooks are excluded. +Reuse the `.env` file described in the AWS environment section if it is already configured. +For an offline-only setup, create a gitignored `.env` file in the repository root with these placeholder values: + +```ini +AWS_DEFAULT_REGION=us-east-1 +AWS_ATHENA_S3_STAGING_DIR=s3://pyathena-offline-placeholder/ +AWS_ATHENA_WORKGROUP=offline +AWS_ATHENA_SPARK_WORKGROUP=offline +AWS_EC2_METADATA_DISABLED=true +``` + +After `just lint`, load `.env` and run: + +```bash +uv run --env-file .env pytest --noconftest -p no:rerunfailures -q \ + tests/pyathena/pandas/test_result_set.py \ + tests/pyathena/polars/test_result_set.py +``` + +The four AWS configuration values are required by `tests/__init__.py`, which pytest still imports with `--noconftest`. +The offline-only `.env` example disables EC2 metadata to prevent implicit credential lookup through that service. +`--noconftest` excludes the AWS session hooks and fixtures; disabling the rerun plugin also avoids its local socket setup in restricted environments. +Use this invocation only for self-contained modules; integration tests need their normal fixtures and a real AWS environment. + +### SQLAlchemy suites + The SQLAlchemy compliance suites under `tests/sqlalchemy/` run with different configurations: use `sqla` for synchronous dialects and `sqla-async` for native asyncio dialects. They do not run PyAthena's own dialect regression tests under `tests/pyathena/sqlalchemy/` and `tests/pyathena/aio/sqlalchemy/`. Run the relevant PyAthena tests too, either through `just test pyathena` or a focused selection during development: @@ -136,17 +168,39 @@ Run the relevant PyAthena tests too, either through `just test pyathena` or a fo uv run --env-file .env pytest -n 1 tests/pyathena/sqlalchemy/ tests/pyathena/aio/sqlalchemy/ -v ``` +### Run tox + To invoke the configured tox environments locally: ```bash uv run --env-file .env just tox ``` +### Record results + Record the tested commit, Python and relevant dependency versions, exact commands, and results in the pull request. Include failed and skipped tests and explain any unrun coverage. Separate real AWS results from mock-based tests and static checks. Sanitize logs before sharing them. +## Organize tests + +Group cursor and engine integration tests in classes, and use standalone functions for stateless helpers. +A unit-test class can group the behavior of one object or common setup, as in `TestTypeSignatureParser`. +Function-oriented utility tests can remain standalone when they use AWS fixtures; `tests/pyathena/pandas/test_util.py` follows this pattern. +SQLAlchemy compliance tests retain the classes, decorators, and plugin setup required by the upstream suite. +Preserve attribution for adapted tests as documented in `NOTICE`. + +Parameter definitions should make inputs, options, and expected behavior visible. +Build mocks, configured result sets, and one-shot readers during fixture setup or test execution. +Pure values and framework type or expression objects can be constructed in parameters. +Choose explicit parameter IDs when generated IDs obscure the case, and keep existing IDs, marks, and fixture scopes when reorganizing tests. + +Use literal expected values for contract-specific behavior. +A library comparison is useful when matching that library is the contract, such as joining pandas chunks versus reading the whole file. +Express intentional differences directly rather than recreating the implementation in the expected-value builder. +Retain dtype and schema checks, option contexts, resource cleanup, and equivalent synchronous and asynchronous scenarios where applicable. + ## GitHub Actions The Test workflow runs for pull requests that change files other than `docs/` and Markdown. diff --git a/tests/pyathena/filesystem/test_s3.py b/tests/pyathena/filesystem/test_s3.py index 9bdd00fd..7565fb5c 100644 --- a/tests/pyathena/filesystem/test_s3.py +++ b/tests/pyathena/filesystem/test_s3.py @@ -4819,9 +4819,26 @@ def test_find_withdirs(self, fs): result = fs.find(dir_, withdirs=False) assert len(result) == 4 # Only files - def test_du(self): - # TODO - pass + def test_du(self, fs): + """Disk usage reports file sizes, their total, and the requested depth.""" + directory = ( + f"s3://{ENV.s3_staging_bucket}/{ENV.s3_staging_key}{ENV.schema}/filesystem/test_du" + ) + first = f"{directory}/first" + second = f"{directory}/nested/second" + try: + fs.pipe_file(first, b"abc") + fs.pipe_file(second, b"12345") + assert fs.du(directory) == 8 + assert fs.du(directory, total=False) == { + fs._strip_protocol(first): 3, + fs._strip_protocol(second): 5, + } + assert fs.du(directory, maxdepth=1) == 3 + assert fs.du(first) == 3 + finally: + with contextlib.suppress(FileNotFoundError): + fs.rm(directory, recursive=True) def test_glob(self, fs): dir_ = f"s3://{ENV.s3_staging_bucket}/{ENV.s3_staging_key}{ENV.schema}/filesystem/test_glob" diff --git a/tests/pyathena/filesystem/test_s3_async.py b/tests/pyathena/filesystem/test_s3_async.py index 19f442c5..d77b3fff 100644 --- a/tests/pyathena/filesystem/test_s3_async.py +++ b/tests/pyathena/filesystem/test_s3_async.py @@ -1775,9 +1775,25 @@ async def test_find_withdirs(self, fs): result = await fs._find(dir_, withdirs=False) assert len(result) == 4 # Only files - def test_du(self): - # TODO - pass + @pytest.mark.asyncio + async def test_du(self, fs): + """Disk usage reports file sizes, their total, and the requested depth.""" + directory = f"s3://{ENV.s3_staging_bucket}/{ENV.s3_staging_key}{ENV.schema}/filesystem/test_async_du" + first = f"{directory}/first" + second = f"{directory}/nested/second" + try: + await fs._pipe_file(first, b"abc") + await fs._pipe_file(second, b"12345") + assert await fs._du(directory) == 8 + assert await fs._du(directory, total=False) == { + fs._strip_protocol(first): 3, + fs._strip_protocol(second): 5, + } + assert await fs._du(directory, maxdepth=1) == 3 + assert await fs._du(first) == 3 + finally: + with contextlib.suppress(FileNotFoundError): + await fs._rm(directory, recursive=True) @pytest.mark.asyncio async def test_glob(self, fs): diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index 0a8cb680..93b2c9c3 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -7,7 +7,7 @@ import csv import io -from unittest.mock import MagicMock, PropertyMock, patch +from unittest.mock import MagicMock, PropertyMock, patch, sentinel import pandas as pd import pyarrow as pa @@ -76,8 +76,8 @@ def test_as_pandas_single_dataframe(self): assert df_iter.as_pandas() is df -_FS = MagicMock(name="pyathena_fs") -_USER_FS = MagicMock(name="user_fs") +_FS = sentinel.pyathena_fs +_USER_FS = sentinel.user_fs class TestAthenaPandasResultSet: @@ -253,28 +253,20 @@ def test_read_parquet_filesystem(self, execute_kwargs, path, filesystem_kwargs): } -def _is_string_dtype(value): - """Return whether a dtype mapping value is a string dtype, ignoring invalid values.""" - try: - dtype = pd.api.types.pandas_dtype(value) - except TypeError: - return False - return isinstance(dtype, pd.StringDtype) or dtype.kind == "U" - - -def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): +def _pyarrow_read_csv_kwargs(types, tab_separated=False, dtype_overrides=None, **kwargs): """Build the pandas.read_csv() options with AthenaPandasResultSet._get_csv_read_options(). Args: types: The Athena types of the result columns, keyed by column name. tab_separated: Whether the result is a tab-separated ``.txt`` file. + dtype_overrides: Entries to add to or replace in the default dtype mapping. + A ``dtype`` in kwargs instead replaces the entire mapping. **kwargs: The pandas.read_csv() options given to ``execute()``. Returns: The options for the PyArrow engine. """ - with patch("pyathena.pandas.result_set.AthenaResultSet.__init__", return_value=None): - result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) + result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) result_set._converter = DefaultPandasTypeConverter() result_set._keep_default_na = False result_set._na_values = ("",) @@ -296,6 +288,8 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): return_value=location, ), ): + if dtype_overrides is not None: + result_set._kwargs["dtype"] = {**result_set.dtypes, **dtype_overrides} assert result_set._reads_csv_with_pyarrow() return result_set._get_csv_read_options("pyarrow", None) @@ -435,127 +429,232 @@ def test_read_csv_without_json_c_converter_keeps_low_memory_default(types, engin assert len(df) == 30 +def _string_series(values, infer_string): + """Express the expected string dtype for the active pandas option.""" + return pd.Series(values, dtype="str" if infer_string else object) + + +def _types_frame(infer_string, parse_time=True): + """Build the typed literal expectation for the all-types CSV.""" + missing = float("nan") + return pd.DataFrame( + { + "ti": pd.Series([1, None], dtype="Int64"), + "si": pd.Series([2, None], dtype="Int64"), + "i": pd.Series([3, None], dtype="Int64"), + "bi": pd.Series([4, None], dtype="Int64"), + "r": pd.Series([1.5, missing], dtype="float64"), + "d": pd.Series([2.25, missing], dtype="float64"), + "c": _string_series(["ab ", missing], infer_string), + "v": _string_series(["plain", missing], infer_string), + "ml": _string_series(['multi\nline "q", x', missing], infer_string), + "arr": _string_series(["[1, 2]", missing], infer_string), + "m": _string_series(["{k=1}", missing], infer_string), + "rw": _string_series(["{a=1, b=x}", missing], infer_string), + "dt": pd.Series([pd.Timestamp("2024-02-29"), pd.NaT], dtype="datetime64[us]"), + "ts": pd.Series( + [pd.Timestamp("2024-02-29 23:59:58.123"), pd.NaT], dtype="datetime64[ns]" + ), + "ts6": pd.Series( + [pd.Timestamp("2024-02-29 23:59:58.123456"), pd.NaT], dtype="datetime64[ns]" + ), + # pandas supplies today's date when it parses a time without a date. + "tm": ( + pd.Series(pd.to_datetime(["12:34:56.789", None])) + if parse_time + else _string_series(["12:34:56.789", None], infer_string) + ), + "iv": _string_series(["2 00:00:00.000", None], infer_string), + "nul": pd.Series([missing, missing], dtype="float64"), + "u": _string_series(["589f6631-9c50-4f58-a121-e2608a04fc64", None], infer_string), + "empty": _string_series([missing, missing], infer_string), + "na": _string_series(["NA", missing], infer_string), + } + ) + + @pytest.mark.filterwarnings("ignore:Could not infer format") @pytest.mark.parametrize("infer_string", [True, False]) @pytest.mark.parametrize( - ("data", "read_csv_kwargs"), + ("data", "types", "read_options", "expected_frame", "pandas_columns"), [ - (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES)), - ( + pytest.param( _TYPES_CSV, - _pyarrow_read_csv_kwargs( - _TYPES, - dtype={ - **_pyarrow_read_csv_kwargs(_TYPES)["dtype"], - "ti": "float32", - "v": "category", - "missing": "int64", - }, - ), + _TYPES, + {}, + _types_frame, + [0, 1, 2, 3, 4, 5, 12, 13, 14, 15, 16, 17, 18], + id="types", + ), + pytest.param( + _TYPES_CSV, + _TYPES, + {"dtype_overrides": {"ti": "float32", "v": "category", "missing": "int64"}}, + lambda infer: _types_frame(infer).astype({"ti": "float32", "v": "category"}), + [0, 1, 2, 3, 4, 5, 7, 12, 13, 14, 15, 16, 17, 18], + id="dtype", ), - (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES, parse_dates=[12, "ts"])), - ( + pytest.param( _TYPES_CSV, - _pyarrow_read_csv_kwargs( - _TYPES, dtype={**_pyarrow_read_csv_kwargs(_TYPES)["dtype"], "dt": "string"} + _TYPES, + {"parse_dates": [12, "ts"]}, + lambda infer: _types_frame(infer, parse_time=False), + [0, 1, 2, 3, 4, 5, 12, 13, 14, 15, 16, 17, 18], + id="parse_dates", + ), + pytest.param( + _TYPES_CSV, + _TYPES, + {"dtype_overrides": {"dt": "string"}}, + lambda infer: _types_frame(infer).assign( + dt=pd.Series(["2024-02-29", None], dtype="string") ), + [0, 1, 2, 3, 4, 5, 12, 13, 14, 15, 16, 17, 18], + id="dtype_of_date_column", ), - ( + pytest.param( '"x","d"\n"1","2024-01-01"\n,\n', - _pyarrow_read_csv_kwargs({"x": "integer", "d": "date"}, dtype={"x": None}), + {"x": "integer", "d": "date"}, + {"dtype": {"x": None}}, + lambda infer: pd.DataFrame( + { + "x": [1.0, float("nan")], + "d": pd.Series([pd.Timestamp("2024-01-01"), pd.NaT], dtype="datetime64[us]"), + } + ), + [0, 1], + id="dtype_none", ), - ( + pytest.param( '"x","x","d"\n"1","2","2024-01-01"\n,,\n', - _pyarrow_read_csv_kwargs({"x": "integer", "d": "date"}), + {"x": "integer", "d": "date"}, + {}, + lambda infer: pd.concat( + [ + pd.Series([1, None], name="x", dtype="Int64"), + pd.Series([2, None], name="x", dtype="Int64"), + pd.Series( + [pd.Timestamp("2024-01-01"), pd.NaT], name="d", dtype="datetime64[us]" + ), + ], + axis=1, + ), + [0, 1, 2], + id="duplicate_names", ), - ( + pytest.param( '"v","n"\n"1","1"\n,\n"nan","3"\n"007","4"\n"1e3","5"\n', - _pyarrow_read_csv_kwargs({"v": "varchar", "n": "integer"}), + {"v": "varchar", "n": "integer"}, + {}, + lambda infer: pd.DataFrame( + { + "v": _string_series(["1", float("nan"), "nan", "007", "1e3"], infer), + "n": pd.Series([1, None, 3, 4, 5], dtype="Int64"), + } + ), + [1], + id="numeric_looking_strings", ), - ( + pytest.param( '"v","w","x"\n"007","a","1"\n,,"2"\n', - _pyarrow_read_csv_kwargs( - {"x": "integer"}, - dtype={ + {"x": "integer"}, + { + "dtype": { "v": pd.ArrowDtype(pa.string()), "w": pd.ArrowDtype(pa.large_string()), "x": pd.Int64Dtype(), - }, + } + }, + lambda infer: pd.DataFrame( + { + "v": pd.Series(["007", None], dtype=pd.ArrowDtype(pa.string())), + "w": pd.Series(["a", None], dtype=pd.ArrowDtype(pa.large_string())), + "x": pd.Series([1, 2], dtype="Int64"), + } ), + [2], + id="arrow_string_dtypes", ), - ( + pytest.param( "001\t2\t003\n004\t5\t\n", - _pyarrow_read_csv_kwargs({"v": "varchar"}, True), + {"v": "varchar"}, + {"tab_separated": True}, + lambda infer: pd.DataFrame( + {"0": [1, 4], "1": [2, 5], "v": _string_series(["3", float("nan")], infer)} + ), + [0, 1], + id="tab_separated_numeric_fields", ), - ( + pytest.param( '"v","n"\n"007","1"\n', - _pyarrow_read_csv_kwargs( - {"v": "varchar", "n": "integer"}, - dtype={**_pyarrow_read_csv_kwargs({"v": "varchar"})["dtype"], 0: str}, - ), + {"v": "varchar", "n": "integer"}, + {"dtype": {"v": str, 0: str}}, + lambda infer: pd.DataFrame({"v": _string_series(["007"], infer), "n": [1]}), + [1], + id="dtype_position_key", ), - ( + pytest.param( '"v"\n"plain"\n"2024-01-01"\n\n', - _pyarrow_read_csv_kwargs({"v": "varchar"}, parse_dates=["v"]), + {"v": "varchar"}, + {"parse_dates": ["v"]}, + lambda infer: pd.DataFrame( + {"v": _string_series(["plain", "2024-01-01", float("nan")], infer)} + ), + [], + id="unparsed_dates", ), - ( + pytest.param( "id \tint \t \nname \tstring \t \n", - _pyarrow_read_csv_kwargs({"col_name": "varchar"}, True), + {"col_name": "varchar"}, + {"tab_separated": True}, + lambda infer: pd.DataFrame( + { + "0": _string_series(["id ", "name "], infer), + "1": _string_series(["int ", "string "], infer), + "col_name": _string_series([" ", " "], infer), + } + ), + [0, 1, 2], + id="tab_separated_extra_fields", ), - ( + pytest.param( "x\t1\t2024-01-01\n\t\t\ny y\t3\t2024-01-02\n", - _pyarrow_read_csv_kwargs({"a": "varchar", "b": "bigint", "c": "date"}, True), + {"a": "varchar", "b": "bigint", "c": "date"}, + {"tab_separated": True}, + lambda infer: pd.DataFrame( + { + "a": _string_series(["x", float("nan"), "y y"], infer), + "b": pd.Series([1, None, 3], dtype="Int64"), + "c": pd.Series( + [pd.Timestamp("2024-01-01"), pd.NaT, pd.Timestamp("2024-01-02")], + dtype="datetime64[us]", + ), + } + ), + [1, 2], + id="tab_separated", ), ], - ids=[ - "types", - "dtype", - "parse_dates", - "dtype_of_date_column", - "dtype_none", - "duplicate_names", - "numeric_looking_strings", - "arrow_string_dtypes", - "tab_separated_numeric_fields", - "dtype_position_key", - "unparsed_dates", - "tab_separated_extra_fields", - "tab_separated", - ], ) -def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_string): - # Without values that cross a read block, the result is the one of - # pandas.read_csv(engine="pyarrow"), except that the columns with a string - # dtype have the values of pandas' C engine, and in a header-less file, its - # missing values. Where the C engine parses a parse_dates column despite its - # string dtype, the PyArrow engine's applying the dtype again is kept. +def test_read_csv_with_pyarrow_matches_pandas( + data, types, read_options, expected_frame, pandas_columns, infer_string +): + """CSV results match literal expectations and pandas where their contracts agree.""" with pd.option_context("future.infer_string", infer_string): - expected = pd.read_csv( - io.BytesIO(data.encode()), - **{**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, - ) - c_engine = pd.read_csv( - io.BytesIO(data.encode()), - **{**read_csv_kwargs, "engine": "c", "dtype": dict(read_csv_kwargs["dtype"])}, - ) - string_columns = { - column for column, value in read_csv_kwargs["dtype"].items() if _is_string_dtype(value) - } - for index, column in enumerate(expected.columns): - if column not in string_columns or c_engine[column].dtype.kind == "M": - continue - if read_csv_kwargs["header"] is None: - # Header-less fields keep the inferred types, and only their missing - # values follow the C engine, which makes extra fields the index. - missing = c_engine[column].isna().to_numpy() - expected.isetitem(index, expected.iloc[:, index].mask(missing, float("nan"))) - else: - expected.isetitem(index, c_engine[column].array) - actual = _read_csv_with_pyarrow( - io.BytesIO(data.encode()), - {**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, - ) - assert_frame_equal(actual, expected, check_exact=True) + read_csv_kwargs = _pyarrow_read_csv_kwargs(types, **read_options) + expected = expected_frame(infer_string) + actual = _read_csv_with_pyarrow(io.BytesIO(data.encode()), read_csv_kwargs) + assert_frame_equal(actual, expected, check_exact=True) + # Explicit positions retain duplicate names and headerless-column parity. + # Mapped string columns use PyAthena's preservation contract instead. + if pandas_columns: + reference = pd.read_csv( + io.BytesIO(data.encode()), + **{**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, + ) + assert_frame_equal( + actual.iloc[:, pandas_columns], reference.iloc[:, pandas_columns], check_exact=True + ) @pytest.mark.parametrize("infer_string", [True, False]) @@ -579,11 +678,10 @@ def test_read_csv_with_pyarrow_ignores_unused_dtype_entries(): dtype={"v": str, "unused": "not-a-dtype", "unsupported": "decimal128(10, 2)[pyarrow]"}, ) data = b'"v"\n"007"\n' - expected = pd.read_csv(io.BytesIO(data), **{**read_csv_kwargs, "dtype": {"v": str}}) actual = _read_csv_with_pyarrow( io.BytesIO(data), {**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])} ) - assert actual.columns.tolist() == expected.columns.tolist() == ["v"] + assert actual.columns.tolist() == ["v"] assert actual["v"].tolist() == ["007"] diff --git a/tests/pyathena/polars/test_result_set.py b/tests/pyathena/polars/test_result_set.py index a39fbcb1..7cec99ea 100644 --- a/tests/pyathena/polars/test_result_set.py +++ b/tests/pyathena/polars/test_result_set.py @@ -258,11 +258,16 @@ def test_read_csv_truncates_timestamps(self, kwargs, expected): class TestPolarsDataFrameIterator: @pytest.mark.parametrize( "reader", - [pl.DataFrame({"a": [1, 2]}), (df for df in [pl.DataFrame({"a": [1]})] * 2)], + ["dataframe", "generator"], ids=["dataframe", "generator"], ) def test_close_stops_iteration(self, reader): """A closed iterator yields nothing for either reader kind.""" + reader = ( + pl.DataFrame({"a": [1, 2]}) + if reader == "dataframe" + else (pl.DataFrame({"a": [1]}) for _ in range(2)) + ) df_iter = PolarsDataFrameIterator(reader, {}, ["a"]) df_iter.close() assert list(df_iter) == []