Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions haystack/components/converters/xlsx.py
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,8 @@ def __init__(
:param table_format: The format to convert the Excel file to.
:param sheet_name: The name of the sheet to read. If None, all sheets are read.
:param read_excel_kwargs: Additional arguments to pass to `pandas.read_excel`.
By default, `keep_default_na` is set to False and `na_values` to `[""]` so literal text
values like "NA" and "N/A" are preserved while empty cells remain empty.
See https://pandas.pydata.org/docs/reference/api/pandas.read_excel.html#pandas-read-excel
:param table_format_kwargs: Additional keyword arguments to pass to the table format function.
- If `table_format` is "csv", these arguments are passed to `pandas.DataFrame.to_csv`.
Expand Down Expand Up @@ -164,6 +166,8 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict]
"""
file_bytes = io.BytesIO(bytestream.data)
resolved_read_excel_kwargs = {
"keep_default_na": False,
"na_values": [""],
**self.read_excel_kwargs,
"sheet_name": self.sheet_name,
"header": None, # Don't assign any pandas column labels
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
---
fixes:
- |
Fixed an issue where ``XLSXToDocument`` dropped literal text values such as ``"NA"`` and ``"N/A"``,
converting them into empty cells. ``keep_default_na=False`` and ``na_values=[""]`` are now passed to
``pandas.read_excel`` by default while preserving truly empty cells and allowing callers to override them
via ``read_excel_kwargs``.
57 changes: 57 additions & 0 deletions test/components/converters/test_xlsx_to_document.py
Original file line number Diff line number Diff line change
Expand Up @@ -254,3 +254,60 @@ def test_link_extraction_from_typed_columns(
]
assert rows == [linked_values + ["1.23"], ["43", "4.5", "False", "2.35"]]
assert documents[0].meta["xlsx"] == {"sheet_name": "Sheet"}

@pytest.mark.parametrize("table_format", ["csv", "markdown"])
def test_run_preserves_literal_na_strings(self, tmp_path: Path, table_format: Literal["csv", "markdown"]) -> None:
workbook = Workbook()
sheet = workbook.active
assert sheet is not None
sheet.append(["Code", "Status", "Blank"])
sheet.append(["NA", "OK", None])
sheet.append(["N/A", "pending", ""])
sheet.append(["North America", "active", ""])
path = tmp_path / "literal_na.xlsx"
workbook.save(path)
workbook.close()

converter = XLSXToDocument(table_format=table_format)
documents = converter.run(sources=[path])["documents"]

assert len(documents) == 1
content = documents[0].content
assert content is not None

if table_format == "csv":
rows = [row[1:] for row in list(csv.reader(io.StringIO(content)))[1:]]
assert rows == [
["Code", "Status", "Blank"],
["NA", "OK", ""],
["N/A", "pending", ""],
["North America", "active", ""],
]
else:
rows = [[cell.strip() for cell in row.split("|")[2:-1]] for row in content.splitlines()[2:]]
assert rows == [
["Code", "Status", "Blank"],
["NA", "OK", ""],
["N/A", "pending", ""],
["North America", "active", ""],
]

def test_run_keep_default_na_override(self, tmp_path: Path) -> None:
workbook = Workbook()
sheet = workbook.active
assert sheet is not None
sheet.append(["Code", "Status"])
sheet.append(["NA", "OK"])
sheet.append(["N/A", "pending"])
path = tmp_path / "override_na.xlsx"
workbook.save(path)
workbook.close()

converter = XLSXToDocument(read_excel_kwargs={"keep_default_na": True})
documents = converter.run(sources=[path])["documents"]

assert len(documents) == 1
content = documents[0].content
assert content is not None
rows = [row[1:] for row in list(csv.reader(io.StringIO(content)))[1:]]
assert rows == [["Code", "Status"], ["", "OK"], ["", "pending"]]