From 6c1e43d5c0d6d6792a225e75155a2b20d2012719 Mon Sep 17 00:00:00 2001 From: LimbC-C <329404569+LimbC-C@users.noreply.github.com> Date: Thu, 17 Sep 2026 09:58:31 +0800 Subject: [PATCH 1/3] fix: preserve row_number columns in CSVToDocument --- .../converters/csvtodocument.mdx | 5 ++++ .../converters/csvtodocument.mdx | 5 ++++ haystack/components/converters/csv.py | 5 ++-- ...sv-row-number-column-7965b5c7df176503.yaml | 12 ++++++++++ .../converters/test_csv_to_document.py | 24 +++++++++++++++++++ 5 files changed, 48 insertions(+), 3 deletions(-) create mode 100644 releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml diff --git a/docs-website/docs/pipeline-components/converters/csvtodocument.mdx b/docs-website/docs/pipeline-components/converters/csvtodocument.mdx index 0320301a199..50f2f1ef53c 100644 --- a/docs-website/docs/pipeline-components/converters/csvtodocument.mdx +++ b/docs-website/docs/pipeline-components/converters/csvtodocument.mdx @@ -29,6 +29,11 @@ Converts CSV files to documents. The component uses UTF-8 encoding by default, but you may specify a different encoding if needed during initialization. You can optionally attach metadata to each document with a `meta` parameter when running the component. +In row mode (`conversion_mode="row"`), pass `content_column` to select the column used as document content. +The remaining columns become metadata, while `meta["row_number"]` records the zero-based row index. +A CSV column named `row_number` is preserved as `csv_row_number`; if that key already exists in the metadata, +a numeric suffix is added, for example `csv_row_number_1`. + ## Usage ### On its own diff --git a/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx b/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx index 0320301a199..50f2f1ef53c 100644 --- a/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx +++ b/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx @@ -29,6 +29,11 @@ Converts CSV files to documents. The component uses UTF-8 encoding by default, but you may specify a different encoding if needed during initialization. You can optionally attach metadata to each document with a `meta` parameter when running the component. +In row mode (`conversion_mode="row"`), pass `content_column` to select the column used as document content. +The remaining columns become metadata, while `meta["row_number"]` records the zero-based row index. +A CSV column named `row_number` is preserved as `csv_row_number`; if that key already exists in the metadata, +a numeric suffix is added, for example `csv_row_number_1`. + ## Usage ### On its own diff --git a/haystack/components/converters/csv.py b/haystack/components/converters/csv.py index 9a4a102da54..f75099dd055 100644 --- a/haystack/components/converters/csv.py +++ b/haystack/components/converters/csv.py @@ -213,9 +213,9 @@ def _build_document_from_row( :param content_column: Column name to use for ``Document.content``. :returns: A ``Document`` with chosen content and merged metadata. Remaining row columns are added to ``meta`` with collision-safe - keys (prefixed with ``csv_`` if needed). + keys (prefixed with ``csv_`` if needed), including a CSV column named ``row_number``. """ - row_meta = dict(base_meta) + row_meta = {**base_meta, "row_number": row_index} # content (strict: content_column must exist; validated by caller) content = self._safe_value(row.get(content_column)) @@ -235,5 +235,4 @@ def _build_document_from_row( suffix += 1 row_meta[key_to_use] = self._safe_value(v) - row_meta["row_number"] = row_index return Document(content=content, meta=row_meta) diff --git a/releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml b/releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml new file mode 100644 index 00000000000..1299a072b83 --- /dev/null +++ b/releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml @@ -0,0 +1,12 @@ +--- +fixes: + - | + ``CSVToDocument`` in row mode now preserves a CSV column named ``row_number`` + as ``csv_row_number`` instead of overwriting its value with the generated row + index. Existing metadata keys are preserved using numeric suffixes when needed. +upgrade: + - | + For CSV inputs with a ``row_number`` column that is not the content column, + the original value is now available in document metadata under + ``csv_row_number`` (or a suffixed key on collision). ``row_number`` continues + to contain the zero-based row index. Other CSV inputs require no changes. diff --git a/test/components/converters/test_csv_to_document.py b/test/components/converters/test_csv_to_document.py index dfb6afcdd65..f9847587477 100644 --- a/test/components/converters/test_csv_to_document.py +++ b/test/components/converters/test_csv_to_document.py @@ -271,3 +271,27 @@ def test_run_utf8_without_bom_is_unchanged(self, tmp_path): assert len(docs) == 1 assert docs[0].content == "Name,City\r\nJosé,München\r\n" + + @pytest.mark.parametrize("meta", [{}, {"csv_row_number": "existing", "csv_row_number_1": "also existing"}]) + def test_row_mode_preserves_row_number_column(self, meta: dict[str, str]) -> None: + source = ByteStream(data=b"text,row_number\nfirst,record-42\nsecond,record-43\n") + converter = CSVToDocument(conversion_mode="row") + + documents = converter.run(sources=[source], content_column="text", meta=meta)["documents"] + + assert [document.content for document in documents] == ["first", "second"] + column_key = "csv_row_number_2" if meta else "csv_row_number" + assert [document.meta for document in documents] == [ + {**meta, "row_number": 0, column_key: "record-42"}, + {**meta, "row_number": 1, column_key: "record-43"}, + ] + + def test_row_mode_row_number_as_content_column(self) -> None: + source = ByteStream(data=b"row_number,author\nrecord-42,Ada\n") + converter = CSVToDocument(conversion_mode="row") + + documents = converter.run(sources=[source], content_column="row_number")["documents"] + + assert len(documents) == 1 + assert documents[0].content == "record-42" + assert documents[0].meta == {"author": "Ada", "row_number": 0} From 38e8f90246724a9984e372819bfe5cc2e24b49dd Mon Sep 17 00:00:00 2001 From: anakin87 Date: Fri, 2 Oct 2026 15:19:05 +0200 Subject: [PATCH 2/3] simplify --- .../converters/csvtodocument.mdx | 6 +- .../converters/csvtodocument.mdx | 5 -- haystack/components/converters/csv.py | 2 +- ...sv-row-number-column-7965b5c7df176503.yaml | 12 +-- .../converters/test_csv_to_document.py | 76 ++++++++----------- 5 files changed, 37 insertions(+), 64 deletions(-) diff --git a/docs-website/docs/pipeline-components/converters/csvtodocument.mdx b/docs-website/docs/pipeline-components/converters/csvtodocument.mdx index 50f2f1ef53c..b580aa8ec22 100644 --- a/docs-website/docs/pipeline-components/converters/csvtodocument.mdx +++ b/docs-website/docs/pipeline-components/converters/csvtodocument.mdx @@ -29,10 +29,8 @@ Converts CSV files to documents. The component uses UTF-8 encoding by default, but you may specify a different encoding if needed during initialization. You can optionally attach metadata to each document with a `meta` parameter when running the component. -In row mode (`conversion_mode="row"`), pass `content_column` to select the column used as document content. -The remaining columns become metadata, while `meta["row_number"]` records the zero-based row index. -A CSV column named `row_number` is preserved as `csv_row_number`; if that key already exists in the metadata, -a numeric suffix is added, for example `csv_row_number_1`. +In row mode (`conversion_mode="row"`), `content_column` selects the column used as document content. +The remaining columns become metadata, and `meta["row_number"]` records the zero-based row index. ## Usage diff --git a/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx b/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx index 50f2f1ef53c..0320301a199 100644 --- a/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx +++ b/docs-website/versioned_docs/version-3.1/pipeline-components/converters/csvtodocument.mdx @@ -29,11 +29,6 @@ Converts CSV files to documents. The component uses UTF-8 encoding by default, but you may specify a different encoding if needed during initialization. You can optionally attach metadata to each document with a `meta` parameter when running the component. -In row mode (`conversion_mode="row"`), pass `content_column` to select the column used as document content. -The remaining columns become metadata, while `meta["row_number"]` records the zero-based row index. -A CSV column named `row_number` is preserved as `csv_row_number`; if that key already exists in the metadata, -a numeric suffix is added, for example `csv_row_number_1`. - ## Usage ### On its own diff --git a/haystack/components/converters/csv.py b/haystack/components/converters/csv.py index f75099dd055..65e59242ed7 100644 --- a/haystack/components/converters/csv.py +++ b/haystack/components/converters/csv.py @@ -213,7 +213,7 @@ def _build_document_from_row( :param content_column: Column name to use for ``Document.content``. :returns: A ``Document`` with chosen content and merged metadata. Remaining row columns are added to ``meta`` with collision-safe - keys (prefixed with ``csv_`` if needed), including a CSV column named ``row_number``. + keys (prefixed with ``csv_`` if needed). """ row_meta = {**base_meta, "row_number": row_index} diff --git a/releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml b/releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml index 1299a072b83..b7f981fdae9 100644 --- a/releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml +++ b/releasenotes/notes/preserve-csv-row-number-column-7965b5c7df176503.yaml @@ -1,12 +1,6 @@ --- fixes: - | - ``CSVToDocument`` in row mode now preserves a CSV column named ``row_number`` - as ``csv_row_number`` instead of overwriting its value with the generated row - index. Existing metadata keys are preserved using numeric suffixes when needed. -upgrade: - - | - For CSV inputs with a ``row_number`` column that is not the content column, - the original value is now available in document metadata under - ``csv_row_number`` (or a suffixed key on collision). ``row_number`` continues - to contain the zero-based row index. Other CSV inputs require no changes. + ``CSVToDocument`` in row mode now preserves values from a CSV column named + ``row_number`` as ``csv_row_number`` in metadata (with a suffix on collision), + instead of overwriting them with the generated row index. diff --git a/test/components/converters/test_csv_to_document.py b/test/components/converters/test_csv_to_document.py index f9847587477..2ffd241f93d 100644 --- a/test/components/converters/test_csv_to_document.py +++ b/test/components/converters/test_csv_to_document.py @@ -138,28 +138,38 @@ def test_row_mode_with_content_column(self, tmp_path): assert docs[0].meta["row_number"] == 0 assert os.path.basename(f) == docs[0].meta["file_path"] - def test_row_mode_meta_collision_prefixed(self, tmp_path): - # ByteStream meta has file_path and encoding; CSV also has those columns. - csv_text = "file_path,encoding,comment\r\nrowpath.csv,latin1,ok\r\n" - f = tmp_path / "collide.csv" - f.write_text(csv_text, encoding="utf-8") - bs = ByteStream.from_file_path(f) - bs.meta["file_path"] = str(f) - bs.meta["encoding"] = "utf-8" + def test_row_mode_row_number_as_content_column(self) -> None: + source = ByteStream(data=b"row_number,author\nrecord-42,Ada\n") + converter = CSVToDocument(conversion_mode="row") - conv = CSVToDocument(conversion_mode="row") - out = conv.run(sources=[bs], content_column="comment") - d = out["documents"][0] - # Original meta preserved - assert d.meta["file_path"] == os.path.basename(str(f)) - assert d.meta["encoding"] == "utf-8" - # CSV columns stored with csv_ prefix (no clobber) - assert d.meta["csv_file_path"] == "rowpath.csv" - assert d.meta["csv_encoding"] == "latin1" - # content column isn't duplicated in meta - assert "comment" not in d.meta - assert d.meta["row_number"] == 0 - assert d.content == "ok" + documents = converter.run(sources=[source], content_column="row_number")["documents"] + + assert len(documents) == 1 + assert documents[0].content == "record-42" + assert documents[0].meta == {"author": "Ada", "row_number": 0} + + @pytest.mark.parametrize("column_name", ["file_path", "row_number"]) + def test_row_mode_meta_collision_prefixed(self, tmp_path, column_name: str): + # file_path collides with source metadata; row_number collides with the generated row index. + csv_text = f"{column_name},encoding,comment\r\nsource-value,latin1,ok\r\n" + path = tmp_path / "collide.csv" + path.write_text(csv_text, encoding="utf-8") + source = ByteStream.from_file_path(path) + source.meta["file_path"] = str(path) + source.meta["encoding"] = "utf-8" + converter = CSVToDocument(conversion_mode="row") + + documents = converter.run(sources=[source], content_column="comment")["documents"] + + assert len(documents) == 1 + assert documents[0].content == "ok" + assert documents[0].meta == { + "file_path": "collide.csv", + "encoding": "utf-8", + "row_number": 0, + f"csv_{column_name}": "source-value", + "csv_encoding": "latin1", + } def test_row_mode_meta_collision_multiple_suffixes(self, tmp_path): """ @@ -271,27 +281,3 @@ def test_run_utf8_without_bom_is_unchanged(self, tmp_path): assert len(docs) == 1 assert docs[0].content == "Name,City\r\nJosé,München\r\n" - - @pytest.mark.parametrize("meta", [{}, {"csv_row_number": "existing", "csv_row_number_1": "also existing"}]) - def test_row_mode_preserves_row_number_column(self, meta: dict[str, str]) -> None: - source = ByteStream(data=b"text,row_number\nfirst,record-42\nsecond,record-43\n") - converter = CSVToDocument(conversion_mode="row") - - documents = converter.run(sources=[source], content_column="text", meta=meta)["documents"] - - assert [document.content for document in documents] == ["first", "second"] - column_key = "csv_row_number_2" if meta else "csv_row_number" - assert [document.meta for document in documents] == [ - {**meta, "row_number": 0, column_key: "record-42"}, - {**meta, "row_number": 1, column_key: "record-43"}, - ] - - def test_row_mode_row_number_as_content_column(self) -> None: - source = ByteStream(data=b"row_number,author\nrecord-42,Ada\n") - converter = CSVToDocument(conversion_mode="row") - - documents = converter.run(sources=[source], content_column="row_number")["documents"] - - assert len(documents) == 1 - assert documents[0].content == "record-42" - assert documents[0].meta == {"author": "Ada", "row_number": 0} From 8e93f9adca554c22e1f9e127b4d0d71b8304244e Mon Sep 17 00:00:00 2001 From: anakin87 Date: Fri, 2 Oct 2026 15:35:25 +0200 Subject: [PATCH 3/3] fix test typing --- test/components/converters/test_csv_to_document.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/test/components/converters/test_csv_to_document.py b/test/components/converters/test_csv_to_document.py index 2ffd241f93d..9717c6265d1 100644 --- a/test/components/converters/test_csv_to_document.py +++ b/test/components/converters/test_csv_to_document.py @@ -4,6 +4,7 @@ import logging import os +from pathlib import Path import pytest @@ -149,7 +150,7 @@ def test_row_mode_row_number_as_content_column(self) -> None: assert documents[0].meta == {"author": "Ada", "row_number": 0} @pytest.mark.parametrize("column_name", ["file_path", "row_number"]) - def test_row_mode_meta_collision_prefixed(self, tmp_path, column_name: str): + def test_row_mode_meta_collision_prefixed(self, tmp_path: Path, column_name: str) -> None: # file_path collides with source metadata; row_number collides with the generated row index. csv_text = f"{column_name},encoding,comment\r\nsource-value,latin1,ok\r\n" path = tmp_path / "collide.csv"