Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,9 @@ Converts CSV files to documents.
The component uses UTF-8 encoding by default, but you may specify a different encoding if needed during initialization.
You can optionally attach metadata to each document with a `meta` parameter when running the component.

In row mode (`conversion_mode="row"`), `content_column` selects the column used as document content.
The remaining columns become metadata, and `meta["row_number"]` records the zero-based row index.

## Usage

### On its own
Expand Down
3 changes: 1 addition & 2 deletions haystack/components/converters/csv.py
Original file line number Diff line number Diff line change
Expand Up @@ -215,7 +215,7 @@ def _build_document_from_row(
Remaining row columns are added to ``meta`` with collision-safe
keys (prefixed with ``csv_`` if needed).
"""
row_meta = dict(base_meta)
row_meta = {**base_meta, "row_number": row_index}

# content (strict: content_column must exist; validated by caller)
content = self._safe_value(row.get(content_column))
Expand All @@ -235,5 +235,4 @@ def _build_document_from_row(
suffix += 1
row_meta[key_to_use] = self._safe_value(v)

row_meta["row_number"] = row_index
return Document(content=content, meta=row_meta)
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
---
fixes:
- |
``CSVToDocument`` in row mode now preserves values from a CSV column named
``row_number`` as ``csv_row_number`` in metadata (with a suffix on collision),
instead of overwriting them with the generated row index.
53 changes: 32 additions & 21 deletions test/components/converters/test_csv_to_document.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@

import logging
import os
from pathlib import Path

import pytest

Expand Down Expand Up @@ -138,28 +139,38 @@ def test_row_mode_with_content_column(self, tmp_path):
assert docs[0].meta["row_number"] == 0
assert os.path.basename(f) == docs[0].meta["file_path"]

def test_row_mode_meta_collision_prefixed(self, tmp_path):
# ByteStream meta has file_path and encoding; CSV also has those columns.
csv_text = "file_path,encoding,comment\r\nrowpath.csv,latin1,ok\r\n"
f = tmp_path / "collide.csv"
f.write_text(csv_text, encoding="utf-8")
bs = ByteStream.from_file_path(f)
bs.meta["file_path"] = str(f)
bs.meta["encoding"] = "utf-8"
def test_row_mode_row_number_as_content_column(self) -> None:
source = ByteStream(data=b"row_number,author\nrecord-42,Ada\n")
converter = CSVToDocument(conversion_mode="row")

conv = CSVToDocument(conversion_mode="row")
out = conv.run(sources=[bs], content_column="comment")
d = out["documents"][0]
# Original meta preserved
assert d.meta["file_path"] == os.path.basename(str(f))
assert d.meta["encoding"] == "utf-8"
# CSV columns stored with csv_ prefix (no clobber)
assert d.meta["csv_file_path"] == "rowpath.csv"
assert d.meta["csv_encoding"] == "latin1"
# content column isn't duplicated in meta
assert "comment" not in d.meta
assert d.meta["row_number"] == 0
assert d.content == "ok"
documents = converter.run(sources=[source], content_column="row_number")["documents"]

assert len(documents) == 1
assert documents[0].content == "record-42"
assert documents[0].meta == {"author": "Ada", "row_number": 0}

@pytest.mark.parametrize("column_name", ["file_path", "row_number"])
def test_row_mode_meta_collision_prefixed(self, tmp_path: Path, column_name: str) -> None:
# file_path collides with source metadata; row_number collides with the generated row index.
csv_text = f"{column_name},encoding,comment\r\nsource-value,latin1,ok\r\n"
path = tmp_path / "collide.csv"
path.write_text(csv_text, encoding="utf-8")
source = ByteStream.from_file_path(path)
source.meta["file_path"] = str(path)
source.meta["encoding"] = "utf-8"
converter = CSVToDocument(conversion_mode="row")

documents = converter.run(sources=[source], content_column="comment")["documents"]

assert len(documents) == 1
assert documents[0].content == "ok"
assert documents[0].meta == {
"file_path": "collide.csv",
"encoding": "utf-8",
"row_number": 0,
f"csv_{column_name}": "source-value",
"csv_encoding": "latin1",
}

def test_row_mode_meta_collision_multiple_suffixes(self, tmp_path):
"""
Expand Down
Loading