Skip to content
Merged
6 changes: 4 additions & 2 deletions pyathena/aio/arrow/cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -144,6 +144,8 @@ async def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional execution parameters.
``block_size`` sets the read block size for this query, and
``connect_timeout`` and ``request_timeout`` override the cursor's values.

Returns:
Self reference for method chaining.
Expand Down Expand Up @@ -182,8 +184,8 @@ async def execute(
retry_config=self._retry_config,
unload=self._unload,
unload_location=unload_location,
connect_timeout=self._connect_timeout,
request_timeout=self._request_timeout,
connect_timeout=kwargs.pop("connect_timeout", self._connect_timeout),
request_timeout=kwargs.pop("request_timeout", self._request_timeout),
result_set_type_hints=options.result_set_type_hints,
**kwargs,
)
Expand Down
9 changes: 8 additions & 1 deletion pyathena/aio/pandas/cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -167,6 +167,11 @@ async def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional pandas read_csv/read_parquet parameters.
``engine``, ``chunksize``, ``block_size``, ``cache_type``, ``max_workers``,
and ``auto_optimize_chunksize`` override the cursor's values for this query.
``storage_options`` and, for UNLOAD results, ``filesystem`` replace
PyAthena's S3 filesystem (see
:class:`~pyathena.pandas.result_set.AthenaPandasResultSet`).

Returns:
Self reference for method chaining.
Expand Down Expand Up @@ -213,7 +218,9 @@ async def execute(
block_size=kwargs.pop("block_size", self._block_size),
cache_type=kwargs.pop("cache_type", self._cache_type),
max_workers=kwargs.pop("max_workers", self._max_workers),
auto_optimize_chunksize=self._auto_optimize_chunksize,
auto_optimize_chunksize=kwargs.pop(
"auto_optimize_chunksize", self._auto_optimize_chunksize
),
result_set_type_hints=options.result_set_type_hints,
**kwargs,
)
Expand Down
13 changes: 9 additions & 4 deletions pyathena/aio/polars/cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -151,6 +151,11 @@ async def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional execution parameters passed to Polars read functions.
``block_size``, ``cache_type``, ``max_workers``, and ``chunksize``
override the cursor's values for this query.
Read function arguments replace the ones the result set chooses, such as
``separator``, ``has_header``, ``schema_overrides``, and ``storage_options``
(see :class:`~pyathena.polars.result_set.AthenaPolarsResultSet`).

Returns:
Self reference for method chaining.
Expand Down Expand Up @@ -189,10 +194,10 @@ async def execute(
retry_config=self._retry_config,
unload=self._unload,
unload_location=unload_location,
block_size=self._block_size,
cache_type=self._cache_type,
max_workers=self._max_workers,
chunksize=self._chunksize,
block_size=kwargs.pop("block_size", self._block_size),
cache_type=kwargs.pop("cache_type", self._cache_type),
max_workers=kwargs.pop("max_workers", self._max_workers),
chunksize=kwargs.pop("chunksize", self._chunksize),
result_set_type_hints=options.result_set_type_hints,
**kwargs,
)
Expand Down
4 changes: 3 additions & 1 deletion pyathena/aio/s3fs/cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -147,6 +147,8 @@ async def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional execution parameters.
``block_size`` sets the read block size for this query, and
``csv_reader`` overrides the cursor's value.

Returns:
Self reference for method chaining.
Expand Down Expand Up @@ -182,7 +184,7 @@ async def execute(
query_execution=query_execution,
arraysize=self.arraysize,
retry_config=self._retry_config,
csv_reader=self._csv_reader,
csv_reader=kwargs.pop("csv_reader", self._csv_reader),
filesystem_class=AioS3FileSystem,
result_set_type_hints=options.result_set_type_hints,
**kwargs,
Expand Down
6 changes: 4 additions & 2 deletions pyathena/arrow/async_cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -171,8 +171,8 @@ def _collect_result_set(
retry_config=self._retry_config,
unload=self._unload,
unload_location=unload_location,
connect_timeout=self._connect_timeout,
request_timeout=self._request_timeout,
connect_timeout=kwargs.pop("connect_timeout", self._connect_timeout),
request_timeout=kwargs.pop("request_timeout", self._request_timeout),
result_set_type_hints=result_set_type_hints,
**kwargs,
)
Expand Down Expand Up @@ -212,6 +212,8 @@ def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional execution parameters.
``block_size`` sets the read block size for this query, and
``connect_timeout`` and ``request_timeout`` override the cursor's values.

Returns:
Tuple of (query_id, future) where future resolves to AthenaArrowResultSet.
Expand Down
6 changes: 4 additions & 2 deletions pyathena/arrow/cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -170,6 +170,8 @@ def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional execution parameters.
``block_size`` sets the read block size for this query, and
``connect_timeout`` and ``request_timeout`` override the cursor's values.

Returns:
Self reference for method chaining.
Expand Down Expand Up @@ -210,8 +212,8 @@ def execute(
retry_config=self._retry_config,
unload=self._unload,
unload_location=unload_location,
connect_timeout=self._connect_timeout,
request_timeout=self._request_timeout,
connect_timeout=kwargs.pop("connect_timeout", self._connect_timeout),
request_timeout=kwargs.pop("request_timeout", self._request_timeout),
result_set_type_hints=options.result_set_type_hints,
**kwargs,
)
Expand Down
22 changes: 22 additions & 0 deletions pyathena/pandas/async_cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,9 @@ def __init__(
chunksize: int | None = None,
result_reuse_enable: bool = False,
result_reuse_minutes: int = CursorIterator.DEFAULT_RESULT_REUSE_MINUTES,
block_size: int | None = None,

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Self-review round one (behavior, failure paths, data/resource boundaries, simplicity and test quality) — FINDINGS
Scope: git diff 785be364e380a7088b82ecf3b5bcb3a5bf01b9ca..bb006761a94daa69f858347479d6d5b03934e295 — all 12 files: the six pandas/Polars cursors and their test files.

Finding (repaired in d542edb): the new block_size and cache_type parameters were inserted before result_reuse_enable, so a caller passing result_reuse_enable or result_reuse_minutes positionally to AsyncPandasCursor(...) would have set block_size/cache_type instead. All three new parameters are now appended after result_reuse_minutes; keyword callers (including connect(cursor_kwargs=...)) are unaffected either way.

Checked without findings:

  • Each changed kwargs.pop() reads a per-call dict (execute()'s own **kwargs, or the dict submitted to the executor for AsyncPandasCursor/AsyncPolarsCursor), so popping does not leak into later calls.
  • The popped keys reach the same explicit result-set parameters they reached through **kwargs before, so per-call values that already worked (e.g. AsyncPandasCursor.execute(block_size=...)) behave the same.
  • AsyncPandasCursor/AsyncPolarsCursor: max_workers given to execute() sets only the S3 read workers; the executor keeps the constructor size.
  • execute(block_size=None) now overrides the cursor value with None, matching the existing PandasCursor pop semantics.
  • Tests: the offline test_read_options cases fail on the base (5 TypeErrors, plus the AsyncPandasCursor constructor case) and assert the actual result-set arguments; the live test_auto_optimize_chunksize covers the real chunked read through cursor_kwargs.

cache_type: str | None = None,
auto_optimize_chunksize: bool = False,
**kwargs,
) -> None:
"""Initialize an AsyncPandasCursor.
Expand All @@ -99,8 +102,13 @@ def __init__(
unload: Whether to wrap queries in ``UNLOAD`` and read the Parquet output.
engine: Parsing engine (``auto``, ``c``, ``python``, or ``pyarrow``).
chunksize: Number of rows per DataFrame chunk when reading CSV results.
If set, it takes precedence over ``auto_optimize_chunksize``.
result_reuse_enable: Whether to enable Athena query result reuse.
result_reuse_minutes: Maximum age of a reused query result in minutes.
block_size: Default block size of the S3 filesystem that reads the results.
cache_type: Default cache type of the S3 filesystem that reads the results.
auto_optimize_chunksize: Whether to choose a chunk size from the size of the
CSV result file when ``chunksize`` is None.
**kwargs: Other cursor arguments, such as ``connection`` and ``converter``,
passed to ``AsyncCursor.__init__``.
"""
Expand All @@ -122,6 +130,9 @@ def __init__(
self._unload = unload
self._engine = engine
self._chunksize = chunksize
self._block_size = block_size
self._cache_type = cache_type
self._auto_optimize_chunksize = auto_optimize_chunksize

@staticmethod
@override
Expand Down Expand Up @@ -170,6 +181,11 @@ def _collect_result_set(
unload_location=unload_location,
engine=kwargs.pop("engine", self._engine),
chunksize=kwargs.pop("chunksize", self._chunksize),
block_size=kwargs.pop("block_size", self._block_size),
cache_type=kwargs.pop("cache_type", self._cache_type),
auto_optimize_chunksize=kwargs.pop(
"auto_optimize_chunksize", self._auto_optimize_chunksize
),
result_set_type_hints=result_set_type_hints,
**kwargs,
)
Expand Down Expand Up @@ -215,6 +231,12 @@ def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional pandas read_csv/read_parquet parameters.
``engine``, ``chunksize``, ``block_size``, ``cache_type``, and
``auto_optimize_chunksize`` override the cursor's values for this query.
``storage_options`` and, for UNLOAD results, ``filesystem`` replace
PyAthena's S3 filesystem (see
:class:`~pyathena.pandas.result_set.AthenaPandasResultSet`).
``max_workers`` sets the number of S3 read workers for this query.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Self-review round two (claims, existing callers, documentation, evidence) — FINDINGS (PR description only; no code change)
Scope: git diff 785be364e380a7088b82ecf3b5bcb3a5bf01b9ca..d542edb21cef28aab2636523585efd1aee5df814 (full pass), PR body, commit messages, changed docstrings, docs/pandas.md, docs/polars.md, docs/aio.md.

Claims checked:

  • "Constructor values were silently lost": on the base the offline constructor case fails with KeyError: 'block_size' (the value never reaches AthenaPandasResultSet); BaseCursor.__init__ documents **kwargs: Ignored (pyathena/common.py).
  • "Per-call values raised TypeError": on the base, PandasCursor/PolarsCursor/AsyncPolarsCursor fail with got multiple values for keyword argument, and AioPandasCursor/AioPolarsCursor fail the same way inside asyncio.to_thread().
  • "max_workers sizes the thread pool" (this docstring): pyathena/async_cursor.py:112-113 uses it for ThreadPoolExecutor, so the per-call value only sets S3 read workers.
  • "auto_optimize_chunksize ... CSV result file": _auto_determine_chunksize is called only from _read_csv (pyathena/pandas/result_set.py:560-562).
  • Existing callers: the round-one parameter-order repair keeps positional calls; per-call keys that already worked keep reaching the same result-set parameters; no other per-call key changes destination.
  • Docs: no statement in docs/ says AsyncPandasCursor lacks these options or that per-call overrides fail; no doc change needed.

Findings (repaired in the PR description):

  1. The live-test claim was overstated: -k filtered the aio files down to the offline test_read_options, so the Polars and aio execute paths had no live run. Ran at the head: both aio files in full (36 passed) and -k chunk over the Polars cursor files (15 passed).
  2. "Tested commit" named bb00676 after the round-one repair; it now names the head and attributes the pandas live run to bb00676.

Limits: the rest of the pandas/Polars suites and just test pyathena run only in CI after Ready.


Returns:
Tuple of (query_id, future) where future resolves to AthenaPandasResultSet.
Expand Down
9 changes: 8 additions & 1 deletion pyathena/pandas/cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -193,6 +193,11 @@ def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional pandas read_csv/read_parquet parameters.
``engine``, ``chunksize``, ``block_size``, ``cache_type``, ``max_workers``,
and ``auto_optimize_chunksize`` override the cursor's values for this query.
``storage_options`` and, for UNLOAD results, ``filesystem`` replace
PyAthena's S3 filesystem (see
:class:`~pyathena.pandas.result_set.AthenaPandasResultSet`).

Returns:
Self reference for method chaining.
Expand Down Expand Up @@ -242,7 +247,9 @@ def execute(
block_size=kwargs.pop("block_size", self._block_size),
cache_type=kwargs.pop("cache_type", self._cache_type),
max_workers=kwargs.pop("max_workers", self._max_workers),
auto_optimize_chunksize=self._auto_optimize_chunksize,
auto_optimize_chunksize=kwargs.pop(
"auto_optimize_chunksize", self._auto_optimize_chunksize
),
result_set_type_hints=options.result_set_type_hints,
**kwargs,
)
Expand Down
29 changes: 16 additions & 13 deletions pyathena/pandas/result_set.py
Original file line number Diff line number Diff line change
Expand Up @@ -310,6 +310,10 @@ def __init__(
result_set_type_hints: Athena type signatures for complex-type columns,
keyed by column name (case-insensitive) or zero-based column index.
**kwargs: Additional arguments passed to pandas.read_csv/read_parquet.
A given ``storage_options``, even None, replaces PyAthena's S3 filesystem

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Self-review round two (pandas UNLOAD) — FINDINGS (PR description only)
Claims checked at 24b56a95830cc0eb4eb142600f29093449194453:

  • "A non-empty storage_options raised ValueError" (commit message): reproduced with pandas 3.0.6 — read_parquet("bucket/unload/", engine="pyarrow", filesystem=<fsspec fs>, storage_options={"anon": True}) raises ValueError: storage_options passed with buffer, or non-supported URL (pandas/io/parquet.py _get_path_or_handle).
  • Same call with storage_options=None: no error, PyAthena's filesystem was used and the option ignored.
  • This docstring: CSV results already replace PyAthena's filesystem for a given storage_options, even None (_read_csv()); UNLOAD now follows it; the manifest/schema statement matches _read_parquet_schema().

Finding: the per-call storage_options=None case is a behavior change for UNLOAD (ignored before; pandas now resolves the filesystem itself). It was missing from the PR description; added to the WHAT section and the release-note items.

for reading the result files, and so does ``filesystem`` for UNLOAD results.
The UNLOAD manifest is still read with the connection's S3 client, and the

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Independent review (relayed), pandas UNLOAD change: FINDINGS (P2 pre-existing, P3 docstring) → P3 repaired in c8d4f2d

Reviewer: Codex CLI 0.160.0, model gpt-6-astra, reasoning effort high, sandbox read-only, session 01a1026a-68d1-75d0-8baf-c02e0494558b. Static review only. Snapshot: detached worktree at 24b56a95830cc0eb4eb142600f29093449194453; reviewed 16a15eaf24c96930fc33b3e169f9d55cac4c9b5e..24b56a95830cc0eb4eb142600f29093449194453 (the contract expansion, full review) with the intended behavior, without the PR number, description, or prior findings. Afterwards, the snapshot and the PR worktree were clean at 24b56a95.

Covered (reviewer): keyword forwarding and exception propagation in all three pandas cursors; default and overridden Parquet reads; CSV parity; manifest/schema reads; the locked pandas 3.0.6 and pyarrow 25.0.1 sources; both new test groups. No functional defect in the argument selection; the four non-default mocked cases and both live cases fail on the previous head, and the tests assert the pandas call boundary and the returned records.

Findings:

  • P2, pre-existing: tests/pyathena/pandas/test_result_set.py:93 — run normally, the mocked test still passes through the AWS session hook in tests/pyathena/conftest.py:15. Author disposition: not changed, as for the same finding recorded earlier on this PR — the tests make no AWS calls themselves, follow the existing pattern, and run offline with --noconftest.
  • P3: this docstring said the UNLOAD manifest is read with PyAthena's filesystem; AthenaResultSet._read_data_manifest() (pyathena/result_set.py:777-778) reads it with the connection's S3 client. Author disposition: repaired (this line). The same wrong statement in the commit message of 24b56a95 stays as published; the PR description is corrected.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Self-review of the repair (24b56a95830cc0eb4eb142600f29093449194453..c8d4f2d642bb5e4b6f360844d1ca6cb64d39bc4c, docstring only) — CLEAN
Round one: no code change; just lint (incl. ruff pydocstyle and mypy) passed at c8d4f2d6. Round two: "manifest ... connection's S3 client" matches _read_data_manifest() (self._client.get_object with the result set's retry config); "schema with PyAthena's filesystem" matches _read_parquet_schema() (ParquetDataset(..., filesystem=self._fs)). The PR description now states the same; no other prose in the PR or docs makes the manifest claim.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Independent follow-up (relayed): CLEAN
Reviewer: Codex CLI 0.160.0, model gpt-6-astra, reasoning effort high, sandbox read-only, session 01a1026d-dab5-72f0-bd79-4e457c6cc409. Static review only. Snapshot: detached worktree at c8d4f2d642bb5e4b6f360844d1ca6cb64d39bc4c; reviewed 24b56a95830cc0eb4eb142600f29093449194453..c8d4f2d642bb5e4b6f360844d1ca6cb64d39bc4c with the finding. Result: the repaired docstring matches both implementations (manifest via the connection's S3 client, schema via PyAthena's filesystem); no other inaccuracies. The snapshot and the PR worktree were clean at c8d4f2d6 afterwards.

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Rebase record: c8d4f2d642bb5e4b6f360844d1ca6cb64d39bc4c → b88146b0778819b87aff1daab599e8191d5e9122, onto a73c6db4 (via 8446e872, where master's #1010 and #1031 tests conflicted with this PR's test additions).

  • git range-diff fe21250a..c8d4f2d6 origin/master..b88146b0: commits 2, 4, 5, 6 unchanged; commits 1 and 3 differ only in test files — test_fetch_all_rows in the Arrow, pandas, and Polars cursor tests (kept master's new body and assertions, then this PR's test_read_options), and in the S3FS cursor test both master's test_fetch_all_rows_custom_converter and its imports and this PR's test_read_options and imports are kept. The step from 8446e872 to a73c6db4 applied without conflicts.
  • Upstream source changes checked against this PR's contracts: JSON-as-text conversion (Return Athena JSON values from SQLAlchemy JSON columns and the pandas, Arrow, and S3FS cursors #1010) in the Arrow/Polars/S3FS result sets and AthenaResultSet._json_converters(), the Arrow CSV ignore_empty_lines change (Keep NULL rows of single-column CSV results in ArrowCursor #1031), and the filesystem lookup parameters (Send the lookup parameters of a file with its lookups #1024) do not touch the per-call arguments, storage_options, filesystem, csv_reader, or the timeouts.
  • At 20f85eb6 (series on 8446e872): just lint passed; offline 48 passed; live 33 passed over the conflict areas (one pandas UserWarning from test_fetch_all_rows[default], reproduced on master a73c6db4). At b88146b0: just lint passed; offline 18 passed.

schema with PyAthena's filesystem.
"""
super().__init__(
connection=connection,
Expand Down Expand Up @@ -778,23 +782,22 @@ def _read_parquet(self, engine) -> DataFrame:
self._unload_location = "/".join(self._data_manifest[0].split("/")[:-1]) + "/"

if engine == "pyarrow":
# pyarrow takes the path without the scheme with an fsspec filesystem.
bucket, key = parse_output_location(self._unload_location)
unload_location = f"{bucket}/{key}"
kwargs = {
"use_threads": True,
}
kwargs: dict[str, Any] = {"use_threads": True, **self._kwargs}
# Given storage_options, even None, pandas opens the files itself,
# as for CSV results.
if "filesystem" not in kwargs and "storage_options" not in kwargs:

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Self-review round one (pandas UNLOAD filesystem/storage_options) — CLEAN
The maintainer asked to include the pandas UNLOAD filesystem collision and to apply the CSV rule to storage_options; this expands the contract, so this is a full pass over 16a15eaf24c96930fc33b3e169f9d55cac4c9b5e..24b56a95830cc0eb4eb142600f29093449194453 (4 source files, 2 test files).

Covered:

  • Default path (no filesystem/storage_options given): unchanged call — PyAthena's _fs, scheme-less bucket/key path, use_threads=True, then the per-call arguments (which can still override use_threads).
  • Given filesystem (fsspec or pyarrow): passed as is with the scheme-less path, which both kinds expect. filesystem=None: the s3:// location, so pandas resolves the filesystem.
  • Given storage_options (even None) without filesystem: no filesystem argument and the s3:// location, the same rule as _read_csv() (pyathena/pandas/result_set.py "Given storage_options, even None").
  • Both given: passed together; pandas' own validation then raises (fsspec: ValueError, pyarrow FS: NotImplementedError), surfaced as OperationalError like any read failure.
  • The manifest read and _read_parquet_schema() keep PyAthena's filesystem; they read the same UNLOAD location the connection wrote to.
  • Tests: the four override cases of test_read_parquet_filesystem fail at 16a15eaf24c96930fc33b3e169f9d55cac4c9b5e and assert the exact read_parquet call; the live test_unload_with_filesystem_options reads a real UNLOAD result with each option, using skip_instance_cache so no filesystem is left in fsspec's cache.

kwargs["filesystem"] = self._fs
if kwargs.get("filesystem") is None:
unload_location = self._unload_location
else:
# pyarrow takes the path without the scheme with a filesystem.
bucket, key = parse_output_location(self._unload_location)
unload_location = f"{bucket}/{key}"
else:
raise ProgrammingError("Engine must be `pyarrow`.")
kwargs.update(self._kwargs)

try:
return pd.read_parquet(
unload_location,
engine=self._engine,
filesystem=self._fs,
**kwargs,
)
return pd.read_parquet(unload_location, engine=self._engine, **kwargs)
except Exception as e:
_logger.exception(f"Failed to read {self.output_location}.")
raise OperationalError(*e.args) from e
Expand Down
13 changes: 9 additions & 4 deletions pyathena/polars/async_cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -183,10 +183,10 @@ def _collect_result_set(
retry_config=self._retry_config,
unload=self._unload,
unload_location=unload_location,
block_size=self._block_size,
cache_type=self._cache_type,
max_workers=self._max_workers,
chunksize=self._chunksize,
block_size=kwargs.pop("block_size", self._block_size),
cache_type=kwargs.pop("cache_type", self._cache_type),
max_workers=kwargs.pop("max_workers", self._max_workers),
chunksize=kwargs.pop("chunksize", self._chunksize),
result_set_type_hints=result_set_type_hints,
**kwargs,
)
Expand Down Expand Up @@ -230,6 +230,11 @@ def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional execution parameters passed to Polars read functions.
``block_size``, ``cache_type``, ``max_workers``, and ``chunksize``
override the cursor's values for this query.
Read function arguments replace the ones the result set chooses, such as
``separator``, ``has_header``, ``schema_overrides``, and ``storage_options``
(see :class:`~pyathena.polars.result_set.AthenaPolarsResultSet`).

Returns:
Tuple of (query_id, future) where future resolves to AthenaPolarsResultSet.
Expand Down
13 changes: 9 additions & 4 deletions pyathena/polars/cursor.py
Original file line number Diff line number Diff line change
Expand Up @@ -189,6 +189,11 @@ def execute(
:class:`~pyathena.options.ExecuteOptions` instance. Individual
keyword arguments take precedence over ``options`` fields.
**kwargs: Additional execution parameters passed to Polars read functions.
``block_size``, ``cache_type``, ``max_workers``, and ``chunksize``
override the cursor's values for this query.
Read function arguments replace the ones the result set chooses, such as
``separator``, ``has_header``, ``schema_overrides``, and ``storage_options``
(see :class:`~pyathena.polars.result_set.AthenaPolarsResultSet`).

Returns:
Self reference for method chaining.
Expand Down Expand Up @@ -229,10 +234,10 @@ def execute(
retry_config=self._retry_config,
unload=self._unload,
unload_location=unload_location,
block_size=self._block_size,
cache_type=self._cache_type,
max_workers=self._max_workers,
chunksize=self._chunksize,
block_size=kwargs.pop("block_size", self._block_size),

Copy link
Copy Markdown
Member Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Self-review round one — out-of-scope note (not changed in this PR): the same "cursor attribute plus **kwargs" collision exists outside the pandas/Polars scope agreed for this PR:

  • pyathena/arrow/cursor.py:213-214, pyathena/arrow/async_cursor.py:174-175, pyathena/aio/arrow/cursor.py:185-186: per-call connect_timeout/request_timeout raise TypeError.
  • pyathena/s3fs/cursor.py:206, pyathena/s3fs/async_cursor.py:175, pyathena/aio/s3fs/cursor.py:185: per-call csv_reader raises TypeError.
  • pyathena/polars/result_set.py:458,488,506,650,681: a per-call storage_options collides with the result set's own storage_options at read time.
    To be filed as a separate issue after the maintainer confirms.

cache_type=kwargs.pop("cache_type", self._cache_type),
max_workers=kwargs.pop("max_workers", self._max_workers),
chunksize=kwargs.pop("chunksize", self._chunksize),
result_set_type_hints=options.result_set_type_hints,
**kwargs,
)
Expand Down
Loading
Loading