Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions haystack/components/converters/xlsx.py
Original file line number Diff line number Diff line change
Expand Up @@ -164,6 +164,16 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict]
"""
file_bytes = io.BytesIO(bytestream.data)
resolved_read_excel_kwargs = {
# Text values that pandas treats as NA by default ("NA", "N/A", "nan",
# "null", ...) are real cell contents, such as a region code or an explicit
# status. Parsing them as NA destroys them: the cell reads as blank in the
# document and the original text cannot be recovered.
"keep_default_na": False,
# Only genuinely empty cells count as missing, so `missingval` and the rest of
# the missing-value handling still apply to blank cells.
"na_values": [""],
# Callers can restore pandas' default NA-string parsing with
# `read_excel_kwargs={"keep_default_na": True, "na_values": [...]}`.
**self.read_excel_kwargs,
"sheet_name": self.sheet_name,
"header": None, # Don't assign any pandas column labels
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
---
fixes:
- |
Fixed ``XLSXToDocument`` discarding cells whose text pandas parses as a
missing value by default. A cell containing ``NA``, ``N/A``, ``nan``,
``null`` or one of the other default NA strings was silently turned into an
empty cell, so text such as a region code or an explicit status could not be
recovered from the resulting Document. Only genuinely empty cells are treated
as missing now, which means ``table_format_kwargs={"missingval": ...}`` still
applies to blank cells. Pass ``read_excel_kwargs={"keep_default_na": True}``
to restore the previous behaviour.
44 changes: 44 additions & 0 deletions test/components/converters/test_xlsx_to_document.py
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,50 @@ def test_run_markdown_missing_value(self, test_files_path: Path) -> None:
== "| | A | B |\n|---:|:------|:------|\n| 1 | col_c | col_d |\n| 2 | True | N/A |"
)

@staticmethod
def _literal_na_workbook() -> bytes:
"""Cells holding the literal text "NA"/"N/A" next to a genuinely empty cell."""
book = Workbook()
sheet = book.active
sheet.append(["Code", "Status"])
sheet.append(["NA", "OK"])
sheet.append(["N/A", "pending"])
sheet.append([None, "blank"])
buffer = io.BytesIO()
book.save(buffer)
return buffer.getvalue()

def test_run_preserves_literal_na_text(self) -> None:
""" "NA"/"N/A" are real values, so they must not be parsed as missing data."""
converter = XLSXToDocument()
content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content
assert content == ",A,B\n1,Code,Status\n2,NA,OK\n3,N/A,pending\n4,,blank\n"

def test_run_preserves_literal_na_text_in_markdown(self) -> None:
converter = XLSXToDocument(table_format="markdown")
content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content
assert content == (
"| | A | B |\n|---:|:-----|:--------|\n| 1 | Code | Status |\n"
"| 2 | NA | OK |\n| 3 | N/A | pending |\n| 4 | | blank |"
)

def test_run_keeps_genuinely_empty_cells_missing(self) -> None:
"""Only empty cells are missing, so `missingval` still reaches the blank cell."""
converter = XLSXToDocument(table_format="markdown", table_format_kwargs={"missingval": "MISSING"})
content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content
# The empty cell in row 4 renders as the configured missing value, while the
# literal "NA" and "N/A" cells keep their text.
assert (
content == "| | A | B |\n|---:|:--------|:--------|\n| 1 | Code | Status |\n"
"| 2 | NA | OK |\n| 3 | N/A | pending |\n| 4 | MISSING | blank |"
)

def test_read_excel_kwargs_can_restore_default_na_parsing(self) -> None:
"""`read_excel_kwargs` still wins, so callers can opt back into pandas' defaults."""
converter = XLSXToDocument(read_excel_kwargs={"keep_default_na": True})
content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content
assert content == ",A,B\n1,Code,Status\n2,,OK\n3,,pending\n4,,blank\n"

@pytest.mark.parametrize(
"sheet_name, expected_sheet_name, expected_content",
[
Expand Down