From 58ba902ac546a00775abe00eaad76833c1de76d6 Mon Sep 17 00:00:00 2001 From: Premvkmishra Date: Fri, 2 Oct 2026 20:03:24 +0530 Subject: [PATCH] Reject negative ignore_rows and ignore_columns in CSVDocumentCleaner --- .../preprocessors/csvdocumentcleaner.mdx | 4 +-- .../preprocessors/csv_document_cleaner.py | 7 +++++ ...gative-ignore-counts-f0ba474431dcc4ed.yaml | 7 +++++ .../test_csv_document_cleaner.py | 26 ++++++++++++++++++- 4 files changed, 41 insertions(+), 3 deletions(-) create mode 100644 releasenotes/notes/csvdocumentcleaner-negative-ignore-counts-f0ba474431dcc4ed.yaml diff --git a/docs-website/docs/pipeline-components/preprocessors/csvdocumentcleaner.mdx b/docs-website/docs/pipeline-components/preprocessors/csvdocumentcleaner.mdx index 95add1767b7..1dd670f2d68 100644 --- a/docs-website/docs/pipeline-components/preprocessors/csvdocumentcleaner.mdx +++ b/docs-website/docs/pipeline-components/preprocessors/csvdocumentcleaner.mdx @@ -28,8 +28,8 @@ Use `CSVDocumentCleaner` to clean CSV documents by removing empty rows and colum ### Parameters -- `ignore_rows`: Number of rows to ignore from the top of the CSV table before processing. If any columns are removed, the same columns will be dropped from the ignored rows. -- `ignore_columns`: Number of columns to ignore from the left of the CSV table before processing. If any rows are removed, the same rows will be dropped from the ignored columns. +- `ignore_rows`: Number of rows to ignore from the top of the CSV table before processing. Must be 0 or greater. If any columns are removed, the same columns will be dropped from the ignored rows. +- `ignore_columns`: Number of columns to ignore from the left of the CSV table before processing. Must be 0 or greater. If any rows are removed, the same rows will be dropped from the ignored columns. - `remove_empty_rows`: Whether to remove entirely empty rows. - `remove_empty_columns`: Whether to remove entirely empty columns. - `keep_id`: Whether to retain the original document ID in the output document. diff --git a/haystack/components/preprocessors/csv_document_cleaner.py b/haystack/components/preprocessors/csv_document_cleaner.py index d8fe240f519..b092cfb19be 100644 --- a/haystack/components/preprocessors/csv_document_cleaner.py +++ b/haystack/components/preprocessors/csv_document_cleaner.py @@ -43,10 +43,17 @@ def __init__( :param remove_empty_rows: Whether to remove rows that are entirely empty. :param remove_empty_columns: Whether to remove columns that are entirely empty. :param keep_id: Whether to retain the original document ID in the output document. + :raises ValueError: If `ignore_rows` or `ignore_columns` is less than 0. Rows and columns ignored using these parameters are preserved in the final output, meaning they are not considered when removing empty rows and columns. """ + if ignore_rows < 0: + raise ValueError("ignore_rows must be greater than or equal to 0.") + + if ignore_columns < 0: + raise ValueError("ignore_columns must be greater than or equal to 0.") + self.ignore_rows = ignore_rows self.ignore_columns = ignore_columns self.remove_empty_rows = remove_empty_rows diff --git a/releasenotes/notes/csvdocumentcleaner-negative-ignore-counts-f0ba474431dcc4ed.yaml b/releasenotes/notes/csvdocumentcleaner-negative-ignore-counts-f0ba474431dcc4ed.yaml new file mode 100644 index 00000000000..dab107c12d5 --- /dev/null +++ b/releasenotes/notes/csvdocumentcleaner-negative-ignore-counts-f0ba474431dcc4ed.yaml @@ -0,0 +1,7 @@ +--- +upgrade: + - | + ``CSVDocumentCleaner`` now raises a ``ValueError`` during initialization if ``ignore_rows`` or ``ignore_columns`` is less than 0. This validation is also enforced when loading serialized pipelines containing negative ignore counts. To migrate, update any configuration using negative counts to use ``0`` or omit the parameter. +fixes: + - | + Fixed an issue in ``CSVDocumentCleaner`` where negative ``ignore_rows`` or ``ignore_columns`` values silently discarded data. diff --git a/test/components/preprocessors/test_csv_document_cleaner.py b/test/components/preprocessors/test_csv_document_cleaner.py index 31195701a71..87291beb9ba 100644 --- a/test/components/preprocessors/test_csv_document_cleaner.py +++ b/test/components/preprocessors/test_csv_document_cleaner.py @@ -2,7 +2,9 @@ # # SPDX-License-Identifier: Apache-2.0 -from haystack import Document +import pytest + +from haystack import Document, default_from_dict from haystack.components.preprocessors.csv_document_cleaner import CSVDocumentCleaner @@ -218,3 +220,25 @@ def test_remove_empty_rows_and_columns_false() -> None: result = csv_document_cleaner.run([csv_document]) cleaned_document = result["documents"][0] assert cleaned_document.content == ",B,C\n,,4\n,,\n" + + +@pytest.mark.parametrize("value", [-1, -5]) +def test_negative_ignore_rows_raises(value: int) -> None: + with pytest.raises(ValueError, match="ignore_rows must be greater than or equal to 0"): + CSVDocumentCleaner(ignore_rows=value) + + +@pytest.mark.parametrize("value", [-1, -5]) +def test_negative_ignore_columns_raises(value: int) -> None: + with pytest.raises(ValueError, match="ignore_columns must be greater than or equal to 0"): + CSVDocumentCleaner(ignore_columns=value) + + +@pytest.mark.parametrize("param", ["ignore_rows", "ignore_columns"]) +def test_negative_ignore_values_raise_from_dict(param: str) -> None: + data = { + "type": "haystack.components.preprocessors.csv_document_cleaner.CSVDocumentCleaner", + "init_parameters": {param: -1}, + } + with pytest.raises(ValueError, match=f"{param} must be greater than or equal to 0"): + default_from_dict(CSVDocumentCleaner, data)