From f1bed4647b34ed377a44a4c8434484c0f1f834d4 Mon Sep 17 00:00:00 2001 From: Lukas Geiger Date: Thu, 1 Oct 2026 17:35:16 +0200 Subject: [PATCH 1/3] fix(export): preserve SQLite sources and publish complete CSV/JSON outputs --- .github/workflows/source-platform-smoke.yml | 2 +- EXPORTFORMAT.md | 21 ++ SQLiteViewer.py | 23 ++- export_atomic.py | 71 +++++++ tests/test_export_transaction.py | 202 ++++++++++++++++++++ 5 files changed, 315 insertions(+), 4 deletions(-) create mode 100644 export_atomic.py create mode 100644 tests/test_export_transaction.py diff --git a/.github/workflows/source-platform-smoke.yml b/.github/workflows/source-platform-smoke.yml index a5028a7..fe56b74 100644 --- a/.github/workflows/source-platform-smoke.yml +++ b/.github/workflows/source-platform-smoke.yml @@ -43,4 +43,4 @@ jobs: - name: Run headless smoke and contract tests env: PYTHONIOENCODING: utf-8 - run: python -m pytest tests/source_platform_smoke.py tests/test_metadata.py -v --basetemp=.pytest_tmp + run: python -m pytest tests tests/source_platform_smoke.py -v --basetemp=.pytest_tmp diff --git a/EXPORTFORMAT.md b/EXPORTFORMAT.md index 43bbae1..4a2a394 100644 --- a/EXPORTFORMAT.md +++ b/EXPORTFORMAT.md @@ -13,6 +13,27 @@ Stand: 2026-05-25 - Der Export ist offline-first und enthält keine Cloud-Synchronisation. - BLOB-Werte werden JSON-kompatibel als Base64-Struktur serialisiert. +## Sichere Veröffentlichung von CSV und JSON + +Der Export hält sichtbare Daten und Herkunft vor dem Dateidialog fest. Als Ziel +sind die offene Hauptdatenbank, per SQL eingebundene Datenbanken und deren +Standardbegleitdateien `-wal`, `-shm` und `-journal` gesperrt. Das gilt auch für +Dateialiase und für Begleitdateien, die noch nicht existieren. Ein Wechsel oder +Schließen der Datenbank während des Dialogs hebt den ursprünglichen Schutz nicht +auf; neu eingebundene Datenbanken werden zusätzlich berücksichtigt. + +Die Ausgabe wird zunächst vollständig in eine eigene temporäre Datei im +Zielverzeichnis geschrieben, synchronisiert und geschlossen. Erst danach wird +das Ziel nach erneuter Schutzprüfung ersetzt. Serialisierungs-, Schreib- oder +Ersetzungsfehler erhalten eine vorhandene Ausgabe; fremde temporäre Dateien +werden nicht entfernt. Fehler bei der Prüfung aktiver Datenbanken brechen den +Export ab. + +Diese Prüfungen garantieren keine Transaktion gegen gleichzeitige externe +Dateisystemänderungen zwischen letzter Prüfung und Ersetzung und keine +Stromausfall-Dauerhaftigkeit. Besondere SQLite-VFS-/Superjournal-Dateien und +App-Einstellungsdateien sind nicht Teil dieses Datenbankschutzvertrags. + ## Struktur ```json diff --git a/SQLiteViewer.py b/SQLiteViewer.py index 38e945e..e037db2 100644 --- a/SQLiteViewer.py +++ b/SQLiteViewer.py @@ -27,6 +27,7 @@ from datetime import datetime from typing import List, Tuple, Any from translator import TranslationSystem +from export_atomic import atomic_export, database_files APP_TITLE = "SQLite Viewer Pro" APP_VERSION = "2.1.0" @@ -992,6 +993,12 @@ def export_csv(self): messagebox.showwarning("Export", action_state["empty_message"]) return + try: + protected = database_files(self) + except Exception as e: + messagebox.showerror("Export-Fehler", str(e)) + return + table = export_context.get("table") or self.table_var.get() or export_context.get("view") or "export" default_name = f"{table}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.csv" @@ -1006,7 +1013,9 @@ def export_csv(self): return try: - with open(path, "w", newline="", encoding="utf-8-sig") as f: + protected |= database_files(self) + with atomic_export(path, protected, newline="", encoding="utf-8-sig", + refresh_sources=lambda: database_files(self)) as f: writer = csv.writer(f, delimiter=";", quoting=csv.QUOTE_MINIMAL) writer.writerow(columns) # Bugsweep 23: BLOB/bytes base64-kodieren, sonst landet ein b'...'-Rohliteral in der @@ -1085,6 +1094,13 @@ def export_json(self): messagebox.showwarning("Export", action_state["empty_message"]) return + try: + protected = database_files(self) + payload = self._build_export_payload() + except Exception as e: + messagebox.showerror("Export-Fehler", str(e)) + return + table = export_context.get("table") or self.table_var.get() or export_context.get("view") or "export" default_name = f"{table}_{datetime.now().strftime('%Y%m%d_%H%M%S')}.json" path = filedialog.asksaveasfilename( @@ -1098,8 +1114,9 @@ def export_json(self): return try: - payload = self._build_export_payload() - with open(path, "w", encoding="utf-8") as handle: + protected |= database_files(self) + with atomic_export(path, protected, encoding="utf-8", + refresh_sources=lambda: database_files(self)) as handle: json.dump(payload, handle, indent=2, ensure_ascii=False) self._set_status(f"Exportiert: {os.path.basename(path)}") diff --git a/export_atomic.py b/export_atomic.py new file mode 100644 index 0000000..47be221 --- /dev/null +++ b/export_atomic.py @@ -0,0 +1,71 @@ +"""Failure-preserving text exports with SQLite source-file protection.""" +from contextlib import contextmanager +import logging +import os +from pathlib import Path +import tempfile + + +def database_files(viewer): + """Capture main and attached database names, including SQLite sidecars.""" + names = [] + if getattr(viewer, 'db_path', None): + names.append(viewer.db_path) + connection = getattr(viewer, 'conn', None) + if connection is not None: + cursor = connection.execute('PRAGMA database_list') + try: + names.extend(row[2] for row in cursor.fetchall() if row[2]) + finally: + cursor.close() + paths = set() + for name in names: + if name == ':memory:': + continue + for path in (Path(name).absolute(), Path(name).resolve()): + paths.add(path) + paths.update(Path(str(path) + suffix) for suffix in ('-wal', '-shm', '-journal')) + return paths + + +def protect_destination(destination, sources): + target = Path(destination).resolve() + for source in sources: + if target == source: + raise ValueError('Das Exportziel ist eine geschützte Datenbankdatei. Bitte einen anderen Pfad wählen.') + try: + same = os.path.samefile(destination, source) + except FileNotFoundError: + same = False + if same: + raise ValueError('Das Exportziel ist eine geschützte Datenbankdatei. Bitte einen anderen Pfad wählen.') + + +@contextmanager +def atomic_export(destination, sources, *, encoding, newline=None, refresh_sources=None): + """Publish a complete sibling temporary file after rechecking identity.""" + destination = Path(destination).absolute() + protect_destination(destination, sources) + temporary = None + try: + with tempfile.NamedTemporaryFile(mode='w', encoding=encoding, newline=newline, + dir=destination.absolute().parent, + prefix='.sqliteviewer-export-', suffix='.tmp', + delete=False) as handle: + temporary = Path(handle.name) + yield handle + handle.flush() + os.fsync(handle.fileno()) + if refresh_sources is not None: + sources = sources | refresh_sources() + protect_destination(destination, sources) + os.replace(temporary, destination) + temporary = None + finally: + if temporary is not None: + try: + temporary.unlink() + except FileNotFoundError: + pass + except OSError: + logging.getLogger(__name__).warning('Cannot remove own export temporary file: %s', temporary, exc_info=True) diff --git a/tests/test_export_transaction.py b/tests/test_export_transaction.py new file mode 100644 index 0000000..023ec70 --- /dev/null +++ b/tests/test_export_transaction.py @@ -0,0 +1,202 @@ +import csv +import json +import os +import sqlite3 +from types import SimpleNamespace + +import pytest +import SQLiteViewer as module +import export_atomic + + +@pytest.fixture +def viewer(tmp_path, monkeypatch): + source = tmp_path / 'source.sqlite' + db = sqlite3.connect(source) + db.execute('create table items(name text)') + db.execute("insert into items values ('Grüße')") + db.commit() + db.close() + conn = sqlite3.connect(source.as_uri() + '?mode=ro', uri=True) + messages = [] + fake = SimpleNamespace(db_path=str(source), conn=conn, current_columns=['name', 'blob'], + current_data=[('Grüße', b'AB')], export_context={'view': 'table', 'table': 'items'}, + table_var=SimpleNamespace(get=lambda: 'items'), _set_status=lambda *_: None) + fake._build_export_payload = lambda: module.SqlViewer._build_export_payload(fake) + for name in ('showerror', 'showinfo', 'showwarning'): + monkeypatch.setattr(module.messagebox, name, lambda *args, name=name: messages.append((name, args))) + yield fake, source, messages + conn.close() + + +def run_export(viewer, kind, destination, monkeypatch, dialog=None): + monkeypatch.setattr(module.filedialog, 'asksaveasfilename', dialog or (lambda **_: str(destination))) + getattr(module.SqlViewer, 'export_' + kind)(viewer) + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +@pytest.mark.parametrize('alias', ['direct', 'hardlink', 'wal', 'shm', 'journal']) +def test_export_rejects_source_and_sidecars(viewer, kind, alias, monkeypatch): + fake, source, messages = viewer + before = source.read_bytes() + target = source + if alias == 'hardlink': + target = source.with_name('alias.out') + os.link(source, target) + elif alias != 'direct': + target = source.with_name(source.name + '-' + alias) + target.write_bytes(b'protected sidecar') + target_before = target.read_bytes() + run_export(fake, kind, target, monkeypatch) + assert source.read_bytes() == before + assert target.read_bytes() == target_before + assert [m[0] for m in messages] == ['showerror'] + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_attached_database_protected(viewer, kind, tmp_path, monkeypatch): + fake, source, messages = viewer + attached = tmp_path / 'attached.sqlite' + db = sqlite3.connect(attached) + db.execute('create table extra(id integer)') + db.close() + fake.conn.execute('ATTACH DATABASE ? AS extra', (str(attached),)) + before = attached.read_bytes() + run_export(fake, kind, attached, monkeypatch) + assert attached.read_bytes() == before + assert messages[0][0] == 'showerror' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_modal_database_change_keeps_original_protected(viewer, kind, monkeypatch): + fake, source, messages = viewer + before = source.read_bytes() + def dialog(**_): + fake.db_path = None + fake.conn.close() + fake.conn = None + return str(source) + run_export(fake, kind, source, monkeypatch, dialog) + assert source.read_bytes() == before + assert messages[0][0] == 'showerror' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +@pytest.mark.parametrize('fault', ['fsync', 'replace', 'serialization']) +def test_failed_export_preserves_previous_output(viewer, kind, fault, tmp_path, monkeypatch): + fake, source, messages = viewer + target = tmp_path / ('output.' + kind) + target.write_bytes(b'previous output') + foreign = tmp_path / '.sqliteviewer-export-foreign.tmp' + foreign.write_bytes(b'foreign') + def fail(*_, **__): + raise OSError('injected failure') + if fault == 'serialization': + class Broken: + def __str__(self): + raise ValueError('invalid value') + fake.current_data = [(Broken(), b'AB')] + else: + monkeypatch.setattr(export_atomic.os, 'fsync' if fault == 'fsync' else 'replace', fail) + run_export(fake, kind, target, monkeypatch) + assert target.read_bytes() == b'previous output' + assert foreign.read_bytes() == b'foreign' + assert list(tmp_path.glob('.sqliteviewer-export-*.tmp')) == [foreign] + assert messages[0][0] == 'showerror' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_export_snapshot_survives_dialog_change(viewer, kind, tmp_path, monkeypatch): + fake, source, messages = viewer + target = tmp_path / ('output.' + kind) + def dialog(**_): + fake.current_columns = ['other'] + fake.current_data = [('changed',)] + fake.db_path = 'different.sqlite' + return str(target) + run_export(fake, kind, target, monkeypatch, dialog) + if kind == 'csv': + with target.open(encoding='utf-8-sig', newline='') as handle: + assert list(csv.reader(handle, delimiter=';')) == [['name', 'blob'], ['Grüße', 'QUI=']] + else: + payload = json.loads(target.read_text(encoding='utf-8')) + assert payload['columns'] == ['name', 'blob'] + assert payload['source']['database_path'] == str(source) + assert payload['result_rows'][0]['name'] == 'Grüße' + assert messages[0][0] == 'showinfo' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_late_source_alias_rejected(viewer, kind, tmp_path, monkeypatch): + fake, source, messages = viewer + target = tmp_path / 'late.out' + before = source.read_bytes() + original = export_atomic.os.fsync + def replace_with_link(fd): + original(fd) + os.link(source, target) + monkeypatch.setattr(export_atomic.os, 'fsync', replace_with_link) + run_export(fake, kind, target, monkeypatch) + assert source.read_bytes() == before + assert target.read_bytes() == before + assert messages[0][0] == 'showerror' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_cancel_creates_nothing(viewer, kind, tmp_path, monkeypatch): + fake, source, messages = viewer + run_export(fake, kind, '', monkeypatch) + assert messages == [] + assert not list(tmp_path.glob('.sqliteviewer-export-*.tmp')) + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_new_attachment_during_serialization_protected(viewer, kind, tmp_path, monkeypatch): + fake, source, messages = viewer + attached = tmp_path / 'new.sqlite' + db = sqlite3.connect(attached) + db.execute('create table items(id integer)') + db.close() + before = attached.read_bytes() + original = export_atomic.os.fsync + def attach(fd): + original(fd) + fake.conn.execute('ATTACH DATABASE ? AS newly_attached', (str(attached),)) + monkeypatch.setattr(export_atomic.os, 'fsync', attach) + run_export(fake, kind, attached, monkeypatch) + assert attached.read_bytes() == before + assert messages[0][0] == 'showerror' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_closed_connection_does_not_bypass_protection(viewer, kind, tmp_path, monkeypatch): + fake, source, messages = viewer + fake.conn.close() + target = tmp_path / 'output' + target.write_bytes(b'previous') + run_export(fake, kind, target, monkeypatch) + assert target.read_bytes() == b'previous' + assert messages[0][0] == 'showerror' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_missing_sidecar_is_reserved(viewer, kind, monkeypatch): + fake, source, messages = viewer + target = source.with_name(source.name + '-wal') + assert not target.exists() + run_export(fake, kind, target, monkeypatch) + assert not target.exists() + assert messages[0][0] == 'showerror' + + +@pytest.mark.parametrize('kind', ['csv', 'json']) +def test_identity_permission_failure_preserves_output(viewer, kind, tmp_path, monkeypatch): + fake, source, messages = viewer + target = tmp_path / 'output' + target.write_bytes(b'previous') + def denied(*_): + raise PermissionError('identity denied') + monkeypatch.setattr(export_atomic.os.path, 'samefile', denied) + run_export(fake, kind, target, monkeypatch) + assert target.read_bytes() == b'previous' + assert messages[0][0] == 'showerror' From d4ec1924cfec4a58f63006baa5722969838b0956 Mon Sep 17 00:00:00 2001 From: Lukas Geiger Date: Thu, 1 Oct 2026 17:37:18 +0200 Subject: [PATCH 2/3] fix(ci): run standalone smoke and full regressions separately --- .github/workflows/source-platform-smoke.yml | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/.github/workflows/source-platform-smoke.yml b/.github/workflows/source-platform-smoke.yml index fe56b74..7e5a64b 100644 --- a/.github/workflows/source-platform-smoke.yml +++ b/.github/workflows/source-platform-smoke.yml @@ -43,4 +43,9 @@ jobs: - name: Run headless smoke and contract tests env: PYTHONIOENCODING: utf-8 - run: python -m pytest tests tests/source_platform_smoke.py -v --basetemp=.pytest_tmp + run: python -m pytest tests/source_platform_smoke.py tests/test_metadata.py -v --basetemp=.pytest_tmp + + - name: Run full regression suite + env: + PYTHONIOENCODING: utf-8 + run: python -m pytest tests -v --basetemp=.pytest_tmp From 5518f2ea03ac6ba64963440855ea89c3e4644b57 Mon Sep 17 00:00:00 2001 From: Lukas Geiger Date: Thu, 1 Oct 2026 17:40:15 +0200 Subject: [PATCH 3/3] fix(export): reject ambiguous Win32 names and preserve lexical reservations --- EXPORTFORMAT.md | 2 ++ export_atomic.py | 8 +++++++- tests/test_export_transaction.py | 24 ++++++++++++++++++++++++ 3 files changed, 33 insertions(+), 1 deletion(-) diff --git a/EXPORTFORMAT.md b/EXPORTFORMAT.md index 4a2a394..e1189b5 100644 --- a/EXPORTFORMAT.md +++ b/EXPORTFORMAT.md @@ -21,6 +21,8 @@ Standardbegleitdateien `-wal`, `-shm` und `-journal` gesperrt. Das gilt auch fü Dateialiase und für Begleitdateien, die noch nicht existieren. Ein Wechsel oder Schließen der Datenbank während des Dialogs hebt den ursprünglichen Schutz nicht auf; neu eingebundene Datenbanken werden zusätzlich berücksichtigt. +Mehrdeutige Windows-Pfadkomponenten mit abschließendem Punkt oder Leerzeichen +werden abgewiesen, bevor Windows den Zielnamen beim Schreiben normalisiert. Die Ausgabe wird zunächst vollständig in eine eigene temporäre Datei im Zielverzeichnis geschrieben, synchronisiert und geschlossen. Erst danach wird diff --git a/export_atomic.py b/export_atomic.py index 47be221..ed2412b 100644 --- a/export_atomic.py +++ b/export_atomic.py @@ -29,9 +29,15 @@ def database_files(viewer): def protect_destination(destination, sources): + lexical_target = Path(os.path.abspath(destination)) + if os.name == 'nt' and any( + part not in ('.', '..') and part.endswith((' ', '.')) + for part in Path(destination).parts + ): + raise ValueError('Das Exportziel enthält einen mehrdeutigen Windows-Dateinamen. Bitte einen anderen Pfad wählen.') target = Path(destination).resolve() for source in sources: - if target == source: + if lexical_target == Path(os.path.abspath(source)) or target == source.resolve(): raise ValueError('Das Exportziel ist eine geschützte Datenbankdatei. Bitte einen anderen Pfad wählen.') try: same = os.path.samefile(destination, source) diff --git a/tests/test_export_transaction.py b/tests/test_export_transaction.py index 023ec70..889f11a 100644 --- a/tests/test_export_transaction.py +++ b/tests/test_export_transaction.py @@ -200,3 +200,27 @@ def denied(*_): run_export(fake, kind, target, monkeypatch) assert target.read_bytes() == b'previous' assert messages[0][0] == 'showerror' + + +@pytest.mark.skipif(os.name != 'nt', reason='Win32 filename normalization') +@pytest.mark.parametrize('kind', ['csv', 'json']) +@pytest.mark.parametrize('suffix', ['.', ' ']) +def test_windows_sidecar_name_normalization_rejected(viewer, kind, suffix, monkeypatch): + fake, source, messages = viewer + reserved = source.with_name(source.name + '-wal') + run_export(fake, kind, str(reserved) + suffix, monkeypatch) + assert not reserved.exists() + assert messages[0][0] == 'showerror' + + +def test_reserved_broken_symlink_name_is_protected(tmp_path, monkeypatch): + source = tmp_path / 'source.sqlite' + reserved = source.with_name(source.name + '-wal') + # Emulate resolution of a broken link without requiring Windows symlink privilege. + foreign = tmp_path / 'missing-foreign' + original = export_atomic.Path.resolve + def resolve(path, *args, **kwargs): + return foreign if path == reserved else original(path, *args, **kwargs) + monkeypatch.setattr(export_atomic.Path, 'resolve', resolve) + with pytest.raises(ValueError, match='geschützte Datenbankdatei'): + export_atomic.protect_destination(reserved, {reserved})