diff --git a/haystack/components/converters/csv.py b/haystack/components/converters/csv.py index 65e59242ed..b9de23a04c 100644 --- a/haystack/components/converters/csv.py +++ b/haystack/components/converters/csv.py @@ -77,6 +77,8 @@ def __init__( self.quotechar = quotechar # Basic validation + if conversion_mode not in ("file", "row"): + raise ValueError(f"CSVToDocument: unsupported conversion_mode {conversion_mode!r}. Choose 'file' or 'row'.") if len(self.delimiter) != 1: raise ValueError("CSVToDocument: delimiter must be a single character.") if len(self.quotechar) != 1: diff --git a/releasenotes/notes/csv-conversion-mode-validation-8f4c2b1d7e9a3056.yaml b/releasenotes/notes/csv-conversion-mode-validation-8f4c2b1d7e9a3056.yaml new file mode 100644 index 0000000000..c5bade60d6 --- /dev/null +++ b/releasenotes/notes/csv-conversion-mode-validation-8f4c2b1d7e9a3056.yaml @@ -0,0 +1,9 @@ +--- +fixes: + - | + ``CSVToDocument`` now rejects an unsupported ``conversion_mode`` in its constructor. + Previously the value was only compared against ``"file"`` in ``run()``, so any other + string, including a typo such as ``"Rows"`` or a value with a trailing space, silently + selected the row mode and produced one ``Document`` per row instead of one per file. + When ``content_column`` was not passed, the raised error also reported + ``conversion_mode='row'`` for a component the caller had never configured that way. diff --git a/test/components/converters/test_csv_to_document.py b/test/components/converters/test_csv_to_document.py index 9717c6265d..663674c6f9 100644 --- a/test/components/converters/test_csv_to_document.py +++ b/test/components/converters/test_csv_to_document.py @@ -201,6 +201,20 @@ def test_init_validates_delimiter_and_quotechar(self): with pytest.raises(ValueError): CSVToDocument(quotechar='""') + def test_init_rejects_an_unknown_conversion_mode(self): + with pytest.raises(ValueError, match="unsupported conversion_mode 'Rows'"): + CSVToDocument(conversion_mode="Rows") # type: ignore[arg-type] + + def test_init_rejects_a_conversion_mode_that_only_differs_in_whitespace_or_case(self): + with pytest.raises(ValueError, match="unsupported conversion_mode 'file '"): + CSVToDocument(conversion_mode="file ") # type: ignore[arg-type] + with pytest.raises(ValueError, match="unsupported conversion_mode ''"): + CSVToDocument(conversion_mode="") # type: ignore[arg-type] + + def test_init_accepts_both_documented_conversion_modes(self): + assert CSVToDocument(conversion_mode="file").conversion_mode == "file" + assert CSVToDocument(conversion_mode="row").conversion_mode == "row" + def test_row_mode_large_file_warns(self, caplog, monkeypatch): # Make the threshold tiny so the warning always triggers. import haystack.components.converters.csv as csv_mod