From e73435f7ffc27966fddb04a0ec48cbbdb156d3aa Mon Sep 17 00:00:00 2001 From: HarshitR2004 Date: Wed, 30 Sep 2026 23:15:04 +0530 Subject: [PATCH] fix(converters): preserve literal NA strings in XLSXToDocument (#12945) --- haystack/components/converters/xlsx.py | 4 ++ ...ps-literal-na-values-b18412ebd88214fe.yaml | 7 +++ .../converters/test_xlsx_to_document.py | 57 +++++++++++++++++++ 3 files changed, 68 insertions(+) create mode 100644 releasenotes/notes/fix-xlsx-drops-literal-na-values-b18412ebd88214fe.yaml diff --git a/haystack/components/converters/xlsx.py b/haystack/components/converters/xlsx.py index 86aeaa0d76..3589e12040 100644 --- a/haystack/components/converters/xlsx.py +++ b/haystack/components/converters/xlsx.py @@ -63,6 +63,8 @@ def __init__( :param table_format: The format to convert the Excel file to. :param sheet_name: The name of the sheet to read. If None, all sheets are read. :param read_excel_kwargs: Additional arguments to pass to `pandas.read_excel`. + By default, `keep_default_na` is set to False and `na_values` to `[""]` so literal text + values like "NA" and "N/A" are preserved while empty cells remain empty. See https://pandas.pydata.org/docs/reference/api/pandas.read_excel.html#pandas-read-excel :param table_format_kwargs: Additional keyword arguments to pass to the table format function. - If `table_format` is "csv", these arguments are passed to `pandas.DataFrame.to_csv`. @@ -164,6 +166,8 @@ def _extract_tables(self, bytestream: ByteStream) -> tuple[list[str], list[dict] """ file_bytes = io.BytesIO(bytestream.data) resolved_read_excel_kwargs = { + "keep_default_na": False, + "na_values": [""], **self.read_excel_kwargs, "sheet_name": self.sheet_name, "header": None, # Don't assign any pandas column labels diff --git a/releasenotes/notes/fix-xlsx-drops-literal-na-values-b18412ebd88214fe.yaml b/releasenotes/notes/fix-xlsx-drops-literal-na-values-b18412ebd88214fe.yaml new file mode 100644 index 0000000000..408e4c3d34 --- /dev/null +++ b/releasenotes/notes/fix-xlsx-drops-literal-na-values-b18412ebd88214fe.yaml @@ -0,0 +1,7 @@ +--- +fixes: + - | + Fixed an issue where ``XLSXToDocument`` dropped literal text values such as ``"NA"`` and ``"N/A"``, + converting them into empty cells. ``keep_default_na=False`` and ``na_values=[""]`` are now passed to + ``pandas.read_excel`` by default while preserving truly empty cells and allowing callers to override them + via ``read_excel_kwargs``. diff --git a/test/components/converters/test_xlsx_to_document.py b/test/components/converters/test_xlsx_to_document.py index 458d564568..19c9a4d866 100644 --- a/test/components/converters/test_xlsx_to_document.py +++ b/test/components/converters/test_xlsx_to_document.py @@ -254,3 +254,60 @@ def test_link_extraction_from_typed_columns( ] assert rows == [linked_values + ["1.23"], ["43", "4.5", "False", "2.35"]] assert documents[0].meta["xlsx"] == {"sheet_name": "Sheet"} + + @pytest.mark.parametrize("table_format", ["csv", "markdown"]) + def test_run_preserves_literal_na_strings(self, tmp_path: Path, table_format: Literal["csv", "markdown"]) -> None: + workbook = Workbook() + sheet = workbook.active + assert sheet is not None + sheet.append(["Code", "Status", "Blank"]) + sheet.append(["NA", "OK", None]) + sheet.append(["N/A", "pending", ""]) + sheet.append(["North America", "active", ""]) + path = tmp_path / "literal_na.xlsx" + workbook.save(path) + workbook.close() + + converter = XLSXToDocument(table_format=table_format) + documents = converter.run(sources=[path])["documents"] + + assert len(documents) == 1 + content = documents[0].content + assert content is not None + + if table_format == "csv": + rows = [row[1:] for row in list(csv.reader(io.StringIO(content)))[1:]] + assert rows == [ + ["Code", "Status", "Blank"], + ["NA", "OK", ""], + ["N/A", "pending", ""], + ["North America", "active", ""], + ] + else: + rows = [[cell.strip() for cell in row.split("|")[2:-1]] for row in content.splitlines()[2:]] + assert rows == [ + ["Code", "Status", "Blank"], + ["NA", "OK", ""], + ["N/A", "pending", ""], + ["North America", "active", ""], + ] + + def test_run_keep_default_na_override(self, tmp_path: Path) -> None: + workbook = Workbook() + sheet = workbook.active + assert sheet is not None + sheet.append(["Code", "Status"]) + sheet.append(["NA", "OK"]) + sheet.append(["N/A", "pending"]) + path = tmp_path / "override_na.xlsx" + workbook.save(path) + workbook.close() + + converter = XLSXToDocument(read_excel_kwargs={"keep_default_na": True}) + documents = converter.run(sources=[path])["documents"] + + assert len(documents) == 1 + content = documents[0].content + assert content is not None + rows = [row[1:] for row in list(csv.reader(io.StringIO(content)))[1:]] + assert rows == [["Code", "Status"], ["", "OK"], ["", "pending"]]