From 6da8638e18a35b04ceb809d17567b9177cfc230c Mon Sep 17 00:00:00 2001 From: makiaveli1 Date: Tue, 29 Sep 2026 06:23:27 +0100 Subject: [PATCH] Fix delimiter detection in rows_from_file() for ragged and single-column files When a CSV or TSV file was not perfectly rectangular, `csv.Sniffer` either raised "Could not determine delimiter" (single-column files) or returned a nonsense delimiter such as "a" for "id,name\n1,Cleo,extra". The latter silently turned the header into a single column called "id,n" and broke the documented extra-field handling. Treat an alphanumeric sniffed delimiter as a failed sniff and fall back to whichever of comma or tab appears on the most lines, using comma when neither appears (a single-column file). Well-formed files are unaffected. --- sqlite_utils/utils.py | 32 ++++++++++++++++++++++--- tests/test_rows_from_file.py | 45 ++++++++++++++++++++++++++++++++++++ 2 files changed, 74 insertions(+), 3 deletions(-) diff --git a/sqlite_utils/utils.py b/sqlite_utils/utils.py index 91fd24ddd..9a7b7b01a 100644 --- a/sqlite_utils/utils.py +++ b/sqlite_utils/utils.py @@ -285,6 +285,20 @@ def _extra_key_strategy( yield row_out +def _fallback_dialect(sample: str) -> type[csv.Dialect]: + """Return comma or tab for a sample that could not be sniffed. + + Whichever appears on the most non-blank lines wins. Comma is used when the + sample contains neither, which is what a single-column file looks like. + """ + lines = [line for line in sample.splitlines() if line.strip()] + if not lines: + return csv.excel + commas = sum(1 for line in lines if "," in line) + tabs = sum(1 for line in lines if "\t" in line) + return csv.excel_tab if tabs > commas else csv.excel + + def rows_from_file( fp: BinaryIO, format: Format | None = None, @@ -389,9 +403,21 @@ class Format(enum.Enum): with buffered: return rows_from_file(buffered, format=Format.JSON) else: - dialect = csv.Sniffer().sniff( - first_bytes.decode(encoding or "utf-8-sig", "ignore") - ) + sample = first_bytes.decode(encoding or "utf-8-sig", "ignore") + try: + dialect = csv.Sniffer().sniff(sample) + except csv.Error: + # Sniffing needs a consistent structure, so it fails on + # single-column and on ragged files. + dialect = _fallback_dialect(sample) + else: + # For ragged input the sniffer does not fail, it returns a + # nonsense delimiter: it reports "a" as the delimiter of + # "id,name\n1,Cleo,extra", which silently mangles the column + # names. A delimiter is punctuation, never a letter or digit, so + # treat an alphanumeric answer as a failed sniff. + if dialect.delimiter.isalnum(): + dialect = _fallback_dialect(sample) rows, _ = rows_from_file( buffered, format=Format.CSV, dialect=dialect, encoding=encoding ) diff --git a/tests/test_rows_from_file.py b/tests/test_rows_from_file.py index a8e7f9d76..20ae25722 100644 --- a/tests/test_rows_from_file.py +++ b/tests/test_rows_from_file.py @@ -131,3 +131,48 @@ def test_detect_format_keeps_streaming_reader_open( assert not buffered_readers[0].closed assert len(list(rows)) == 2000 assert buffered_readers[0].closed + + +@pytest.mark.parametrize( + "input,expected_format,expected", + ( + # csv.Sniffer raises "Could not determine delimiter" on a single-column + # file, so it could not be loaded at all + (b"name\nalpha\nbeta", Format.CSV, [{"name": "alpha"}, {"name": "beta"}]), + # On ragged input the sniffer does not raise, it reports "a" as the + # delimiter, which turned the header into a single column called "id,n" + ( + b"id,name\n1,Cleo\nextra", + Format.CSV, + [{"id": "1", "name": "Cleo"}, {"id": "extra", "name": None}], + ), + # The same, for a tab-separated file + ( + b"id\tname\n1\tCleo\nextra", + Format.TSV, + [{"id": "1", "name": "Cleo"}, {"id": "extra", "name": None}], + ), + ), +) +def test_rows_from_file_falls_back_when_sniffing_fails( + input, expected_format, expected +): + # csv.Sniffer either raises "Could not determine delimiter" (single-column) or + # returns a nonsense alphanumeric delimiter (ragged), which mangled the column + # names of every file that was not perfectly rectangular. + rows, format = rows_from_file(BytesIO(input)) + assert format == expected_format + assert list(rows) == expected + + +def test_rows_from_file_single_column_format_is_csv(): + _rows, format = rows_from_file(BytesIO(b"name\nalpha")) + assert format == Format.CSV + + +def test_rows_from_file_ragged_rows_raise_row_error_in_autodetect(): + # The documented behaviour for rows with more values than headings still holds + # once the delimiter is detected correctly. + with pytest.raises(RowError): + rows, _ = rows_from_file(BytesIO(b"id,name\n1,Cleo,oops")) + list(rows)