diff --git a/haystack/components/converters/xlsx.py b/haystack/components/converters/xlsx.py index 69fdb1b16c..f82225b8f0 100644 --- a/haystack/components/converters/xlsx.py +++ b/haystack/components/converters/xlsx.py @@ -164,6 +164,16 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict] """ file_bytes = io.BytesIO(bytestream.data) resolved_read_excel_kwargs = { + # Text values that pandas treats as NA by default ("NA", "N/A", "nan", + # "null", ...) are real cell contents, such as a region code or an explicit + # status. Parsing them as NA destroys them: the cell reads as blank in the + # document and the original text cannot be recovered. + "keep_default_na": False, + # Only genuinely empty cells count as missing, so `missingval` and the rest of + # the missing-value handling still apply to blank cells. + "na_values": [""], + # Callers can restore pandas' default NA-string parsing with + # `read_excel_kwargs={"keep_default_na": True, "na_values": [...]}`. **self.read_excel_kwargs, "sheet_name": self.sheet_name, "header": None, # Don't assign any pandas column labels diff --git a/releasenotes/notes/xlsx-preserve-literal-na-cell-text-a3f7c9d1e4b80265.yaml b/releasenotes/notes/xlsx-preserve-literal-na-cell-text-a3f7c9d1e4b80265.yaml new file mode 100644 index 0000000000..56468abad2 --- /dev/null +++ b/releasenotes/notes/xlsx-preserve-literal-na-cell-text-a3f7c9d1e4b80265.yaml @@ -0,0 +1,11 @@ +--- +fixes: + - | + Fixed ``XLSXToDocument`` discarding cells whose text pandas parses as a + missing value by default. A cell containing ``NA``, ``N/A``, ``nan``, + ``null`` or one of the other default NA strings was silently turned into an + empty cell, so text such as a region code or an explicit status could not be + recovered from the resulting Document. Only genuinely empty cells are treated + as missing now, which means ``table_format_kwargs={"missingval": ...}`` still + applies to blank cells. Pass ``read_excel_kwargs={"keep_default_na": True}`` + to restore the previous behaviour. diff --git a/test/components/converters/test_xlsx_to_document.py b/test/components/converters/test_xlsx_to_document.py index fa2da6a29f..eabc2316f5 100644 --- a/test/components/converters/test_xlsx_to_document.py +++ b/test/components/converters/test_xlsx_to_document.py @@ -108,6 +108,50 @@ def test_run_markdown_missing_value(self, test_files_path: Path) -> None: == "| | A | B |\n|---:|:------|:------|\n| 1 | col_c | col_d |\n| 2 | True | N/A |" ) + @staticmethod + def _literal_na_workbook() -> bytes: + """Cells holding the literal text "NA"/"N/A" next to a genuinely empty cell.""" + book = Workbook() + sheet = book.active + sheet.append(["Code", "Status"]) + sheet.append(["NA", "OK"]) + sheet.append(["N/A", "pending"]) + sheet.append([None, "blank"]) + buffer = io.BytesIO() + book.save(buffer) + return buffer.getvalue() + + def test_run_preserves_literal_na_text(self) -> None: + """ "NA"/"N/A" are real values, so they must not be parsed as missing data.""" + converter = XLSXToDocument() + content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content + assert content == ",A,B\n1,Code,Status\n2,NA,OK\n3,N/A,pending\n4,,blank\n" + + def test_run_preserves_literal_na_text_in_markdown(self) -> None: + converter = XLSXToDocument(table_format="markdown") + content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content + assert content == ( + "| | A | B |\n|---:|:-----|:--------|\n| 1 | Code | Status |\n" + "| 2 | NA | OK |\n| 3 | N/A | pending |\n| 4 | | blank |" + ) + + def test_run_keeps_genuinely_empty_cells_missing(self) -> None: + """Only empty cells are missing, so `missingval` still reaches the blank cell.""" + converter = XLSXToDocument(table_format="markdown", table_format_kwargs={"missingval": "MISSING"}) + content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content + # The empty cell in row 4 renders as the configured missing value, while the + # literal "NA" and "N/A" cells keep their text. + assert ( + content == "| | A | B |\n|---:|:--------|:--------|\n| 1 | Code | Status |\n" + "| 2 | NA | OK |\n| 3 | N/A | pending |\n| 4 | MISSING | blank |" + ) + + def test_read_excel_kwargs_can_restore_default_na_parsing(self) -> None: + """`read_excel_kwargs` still wins, so callers can opt back into pandas' defaults.""" + converter = XLSXToDocument(read_excel_kwargs={"keep_default_na": True}) + content = converter.run([ByteStream(self._literal_na_workbook())])["documents"][0].content + assert content == ",A,B\n1,Code,Status\n2,,OK\n3,,pending\n4,,blank\n" + @pytest.mark.parametrize( "sheet_name, expected_sheet_name, expected_content", [