Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,17 @@

## [Unreleased]

- **Bugfix-Sweep (2026-10-06)**:
- **FTS5-Delete-Trigger repariert (Schema v5)**: Die Trigger nutzten den FTS5-`'delete'`-Befehl auf regulären FTS5-Tabellen → `SQL logic error` bei jedem Löschen von Chunks (Re-Ingest, `index --force`, Web-Viewer-Löschen, `deduplicate`). Bestehende Datenbanken werden beim Öffnen migriert.
- **Löschen konsistent**: Web-Viewer und `deduplicate` entfernen zuerst die DB-Einträge (Transaktion) und verschieben danach die Datei in `_Papierkorb`; Fehler pro Eintrag brechen den Lauf nicht mehr ab.
- **GUI**: Falsche relative Imports (`..schema`/`..config`) ließen die Dokumentliste immer leer; nach dem Sortieren wurde das falsche Dokument geöffnet; Scan-Threads nutzten eine im GUI-Thread erzeugte SQLite-Connection (jede Datei schlug fehl); EventBus-Handler liefen in Worker-Threads (jetzt Queued-Dispatch in den GUI-Thread).
- **Summarizer**: Fehlgeschlagene Chunks führen zu Status `error` statt `done`; verwaiste `processing`-Einträge (> 30 min) werden zurückgesetzt, Ctrl+C setzt das aktuelle Item auf `pending`; Kostenschätzung auf claude-haiku-4-5-Preise ($1/$5 pro MTok) aktualisiert.
- **Suche**: FTS5-Syntaxfehler (z.B. `COVID-19`, `E-Mail`, `c++`) werden mit quotierten Tokens wiederholt, danach LIKE-Fallback (auch für Dokumente); `%`/`_` werden in LIKE-Fallbacks und Verzeichnisfiltern escaped; Verzeichnisfilter matchen keine Geschwister-Ordner mehr (`docs` ≠ `docs2`).
- **Ingest**: Re-Ingest setzt Zusammenfassungen und Queue-Status zurück; archivierte Dokumente werden über `archived_path` gefunden; UTF-8-BOM wird entfernt (Frontmatter-Erkennung); der Chunker erzwingt die Obergrenze von 500 Wörtern auch bei Text ohne Satzgrenzen.
- **Sonstiges**: `is_active = 0` bleibt beim Skill-Index erhalten; Operator-Präzedenz bei `content_hash` korrigiert; `Config` teilt Default-Listen nicht mehr zwischen Instanzen; `zoll_station.py` ruft die CLI als Modul auf; plattformübergreifendes Öffnen von Dateien (Windows/macOS/Linux).
- **Transit-Sync optional**: `sqlite-transit-sync` ist nicht auf PyPI – Import ist jetzt optional, die Sync-Tests werden ohne Paket übersprungen statt die gesamte Test-Collection abzubrechen.
- **Tests**: 43 neue Regressionstests (`tests/test_bugfixes_2026_10.py`) plus Lifecycle-Tests in `tests/test_transit.py`.

- **Discoverability, Visual Architecture & Level 1 SBOM (Pfad B, 2026-09-22)**:
- **18-Point Bilingual Navigation Parity**: Implemented 18-point dual-anchor navigation parity (`<a id="sec-01"></a>`...`<a id="sec-18"></a>`) across `README.md` and `README_de.md` with reciprocal dual anchors for backward compatibility.
- **Target Personas & SEO Discovery**: Added explicit persona profiles (`[PERSONA-01]` to `[PERSONA-04]`) with high-intent search queries for AI engineers, researchers, compliance officers, and automation builders.
Expand Down
28 changes: 23 additions & 5 deletions chunker.py
Original file line number Diff line number Diff line change
Expand Up @@ -65,7 +65,7 @@ def split_frontmatter(text: str) -> Tuple[Optional[str], str]:
if not text:
return None, ""

text = text.strip()
text = text.strip().lstrip("\ufeff") # UTF-8-BOM wuerde '---' verdecken

# Frontmatter: beginnt mit --- und endet mit ---
if text.startswith("---"):
Expand Down Expand Up @@ -116,6 +116,21 @@ def _split_sentences(text: str) -> List[str]:
return segments


def _hard_split(segments: List[str], window: int) -> List[str]:
"""Erzwingt die harte Obergrenze: Segmente > MAX_CHUNK_SIZE (z.B. lange
Einzeiler ohne Satzgrenzen) werden in Wortfenster der Groesse `window`
zerlegt."""
result: List[str] = []
for segment in segments:
words = segment.split()
if len(words) <= MAX_CHUNK_SIZE:
result.append(segment)
continue
for i in range(0, len(words), window):
result.append(" ".join(words[i:i + window]))
return result


def chunk_text(text: str, chunk_size: int = DEFAULT_CHUNK_SIZE,
overlap: int = DEFAULT_OVERLAP,
separate_frontmatter: bool = True) -> List[Chunk]:
Expand Down Expand Up @@ -151,8 +166,9 @@ def chunk_text(text: str, chunk_size: int = DEFAULT_CHUNK_SIZE,
))
chunk_idx += 1

# Body in Segmente splitten
segments = _split_sentences(body)
# Body in Segmente splitten (Segmente > MAX_CHUNK_SIZE hart teilen)
segments = _hard_split(_split_sentences(body),
max(1, min(chunk_size, MAX_CHUNK_SIZE)))
if not segments:
return chunks

Expand Down Expand Up @@ -225,8 +241,10 @@ def chunk_text(text: str, chunk_size: int = DEFAULT_CHUNK_SIZE,
content = "\n\n".join(current_parts)
tokens = estimate_tokens(content)

# Zu kleiner letzter Chunk? An vorherigen anhaengen
if tokens < MIN_CHUNK_SIZE and len(chunks) > 0 and not chunks[-1].is_frontmatter:
# Zu kleiner letzter Chunk? An vorherigen anhaengen (sofern die
# harte Obergrenze dabei nicht ueberschritten wird)
if (tokens < MIN_CHUNK_SIZE and len(chunks) > 0 and not chunks[-1].is_frontmatter
and chunks[-1].token_count + tokens <= MAX_CHUNK_SIZE):
prev = chunks[-1]
merged = prev.content + "\n\n" + content
chunks[-1] = Chunk(
Expand Down
4 changes: 3 additions & 1 deletion config.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@

__all__ = ["Config", "get_config"]

import copy
import json
import os
from pathlib import Path
Expand Down Expand Up @@ -66,7 +67,8 @@ class Config:
"""JSON-basierte Konfiguration."""

def __init__(self, config_path: Optional[Path] = None):
self._data = dict(DEFAULT_CONFIG)
# deepcopy: Listen (z.B. indexed_directories) nicht mit DEFAULT_CONFIG teilen
self._data = copy.deepcopy(DEFAULT_CONFIG)
self._path = config_path
if config_path and config_path.exists():
self._load(config_path)
Expand Down
107 changes: 75 additions & 32 deletions digest.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,12 +31,29 @@
from .config import get_config, Config
from .ingestor import DocumentIngestor
from .summarizer import Summarizer
from .utils import dir_filter_sql, escape_like, fts5_quote, move_to_trash, resolve_document_path

# Lazy imports fuer optionale Module
_SkillIndexer = None
_WikiIndexer = None


def _fts_fetchall(conn: sqlite3.Connection, sql: str, params: list) -> list:
"""Fuehrt eine FTS5-Abfrage aus (params[0] = MATCH-Query).

Bei FTS5-Syntaxfehlern durch Freitext (z.B. 'COVID-19', 'E-Mail', 'c++',
unbalancierte Anfuehrungszeichen) wird mit quotierten Tokens wiederholt.
Scheitert auch das, wird der Fehler weitergereicht (-> LIKE-Fallback).
"""
try:
return conn.execute(sql, params).fetchall()
except sqlite3.OperationalError:
safe = fts5_quote(params[0])
if not safe or safe == params[0]:
raise
return conn.execute(sql, [safe, *params[1:]]).fetchall()


def _get_skill_indexer():
global _SkillIndexer
if _SkillIndexer is None:
Expand Down Expand Up @@ -190,7 +207,7 @@ def search(self, query: str, *, limit: int = 20,
LIMIT ?
"""

rows = conn.execute(sql, params).fetchall()
rows = _fts_fetchall(conn, sql, params)

results = []
seen_skills = set()
Expand Down Expand Up @@ -225,8 +242,8 @@ def _search_fallback(self, conn: sqlite3.Connection, query: str,
limit: int, skill_type: Optional[str],
category: Optional[str]) -> List[Dict[str, Any]]:
"""LIKE-basierte Fallback-Suche wenn FTS5 fehlschlaegt."""
where_parts = ["(sc.content LIKE ? OR si.skill_name LIKE ?)"]
like = f"%{query}%"
where_parts = ["(sc.content LIKE ? ESCAPE '\\' OR si.skill_name LIKE ? ESCAPE '\\')"]
like = f"%{escape_like(query)}%"
params: list = [like, like]

if skill_type:
Expand Down Expand Up @@ -489,9 +506,10 @@ def get_directories(self) -> List[Dict[str, Any]]:
conn = self._get_conn()
result = []
for d in dirs:
where_sql, where_params = dir_filter_sql("source_dir", d)
count = conn.execute(
"SELECT COUNT(*) FROM documents WHERE source_dir LIKE ?",
(d + "%",)
f"SELECT COUNT(*) FROM documents WHERE {where_sql}",
where_params
).fetchone()[0]
result.append({"path": d, "doc_count": count})
conn.close()
Expand Down Expand Up @@ -611,7 +629,7 @@ def _search_documents(self, query: str, *,
"""FTS5-Suche ueber ingested Dokumente."""
conn = self._get_conn()
try:
rows = conn.execute("""
rows = _fts_fetchall(conn, """
SELECT
d.filename,
d.file_type,
Expand All @@ -625,7 +643,7 @@ def _search_documents(self, query: str, *,
WHERE document_fts MATCH ?
ORDER BY document_fts.rank
LIMIT ?
""", (query, limit)).fetchall()
""", [query, limit])

results = []
seen = set()
Expand All @@ -644,10 +662,36 @@ def _search_documents(self, query: str, *,
})
return results
except Exception:
return []
# FTS5 Fallback: LIKE-Suche
return self._search_documents_fallback(conn, query, limit)
finally:
conn.close()

def _search_documents_fallback(self, conn: sqlite3.Connection, query: str,
limit: int) -> List[Dict[str, Any]]:
"""LIKE-basierte Fallback-Suche ueber Dokumente wenn FTS5 fehlschlaegt."""
if not query.strip():
return []
like = f"%{escape_like(query)}%"
try:
rows = conn.execute("""
SELECT DISTINCT d.filename, d.file_type, d.word_count
FROM documents d
LEFT JOIN document_chunks dc ON dc.doc_id = d.id
WHERE dc.content LIKE ? ESCAPE '\\' OR d.filename LIKE ? ESCAPE '\\'
LIMIT ?
""", (like, like, limit)).fetchall()
return [{
'source': 'document',
'name': r['filename'],
'type': r['file_type'],
'snippet': '(LIKE-Fallback)',
'relevance': 0,
'word_count': r['word_count'],
} for r in rows]
except sqlite3.Error:
return []

# ==================================================================
# WIKI-ARTIKEL INDEXIERUNG
# ==================================================================
Expand Down Expand Up @@ -743,7 +787,7 @@ def search_wikis(self, query: str, *, limit: int = 20,
LIMIT ?
"""

rows = conn.execute(sql, params).fetchall()
rows = _fts_fetchall(conn, sql, params)

results = []
seen_wikis = set()
Expand Down Expand Up @@ -777,8 +821,8 @@ def search_wikis(self, query: str, *, limit: int = 20,
def _search_wikis_fallback(self, conn: sqlite3.Connection, query: str,
limit: int, category: Optional[str]) -> List[Dict[str, Any]]:
"""LIKE-basierte Fallback-Suche wenn FTS5 fehlschlaegt."""
where_parts = ["(wc.content LIKE ? OR wi.title LIKE ?)"]
like = f"%{query}%"
where_parts = ["(wc.content LIKE ? ESCAPE '\\' OR wi.title LIKE ? ESCAPE '\\')"]
like = f"%{escape_like(query)}%"
params: list = [like, like]

if category:
Expand Down Expand Up @@ -1305,14 +1349,13 @@ def main():
print("Keine Duplikate gefunden!")
else:
print(f"{len(dups)} Gruppen von Duplikaten gefunden.\n")
import shutil
import os
trash_dir = kd.db_path.parent / "_Papierkorb"
trash_dir.mkdir(parents=True, exist_ok=True)

for row in dups:
hash_val = row['content_hash']
docs = conn.execute("SELECT id, file_path, filename FROM documents WHERE content_hash=? ORDER BY id", (hash_val,)).fetchall()
docs = conn.execute("SELECT id, file_path, filename, archived_path FROM documents WHERE content_hash=? ORDER BY id", (hash_val,)).fetchall()
print(f"\n--- Duplikat-Gruppe ({len(docs)} Dateien) ---")
for i, d in enumerate(docs):
print(f" {i+1}: {d['file_path']}")
Expand All @@ -1324,28 +1367,28 @@ def main():

for d in docs[1:]:
doc_id = d['id']
file_path = d['file_path']
file_path = resolve_document_path(d)
print(f"-> Loesche: {file_path}")

if os.path.exists(file_path):
base_name = os.path.basename(file_path)
trash_path = trash_dir / base_name
counter = 1
while trash_path.exists():
name, ext = os.path.splitext(base_name)
trash_path = trash_dir / f"{name}_{counter}{ext}"
counter += 1

# Erst DB-Eintrag in einer Transaktion entfernen, dann
# die Datei verschieben: schlaegt das Verschieben fehl,
# bleibt die DB trotzdem konsistent.
try:
with conn:
conn.execute("DELETE FROM summaries WHERE source_type='document' AND source_id=?", (doc_id,))
conn.execute("DELETE FROM document_keywords WHERE doc_id=?", (doc_id,))
conn.execute("DELETE FROM document_chunks WHERE doc_id=?", (doc_id,))
conn.execute("DELETE FROM digest_queue WHERE source_type='document' AND source_id=?", (doc_id,))
conn.execute("DELETE FROM documents WHERE id=?", (doc_id,))
except sqlite3.Error as e:
print(f" Fehler beim Loeschen aus der DB (Datei unveraendert): {e}")
continue

if file_path and os.path.exists(file_path):
try:
shutil.move(file_path, trash_path)
move_to_trash(file_path, trash_dir)
except Exception as e:
print(f" Fehler beim Verschieben: {e}")

conn.execute("DELETE FROM summaries WHERE source_type='document' AND source_id=?", (doc_id,))
conn.execute("DELETE FROM document_keywords WHERE doc_id=?", (doc_id,))
conn.execute("DELETE FROM document_chunks WHERE doc_id=?", (doc_id,))
conn.execute("DELETE FROM digest_queue WHERE source_type='document' AND source_id=?", (doc_id,))
conn.execute("DELETE FROM documents WHERE id=?", (doc_id,))
conn.commit()
print(f" Fehler beim Verschieben (DB-Eintrag bereits entfernt): {e}")
print("\nDuplikat-Bereinigung abgeschlossen!")
finally:
conn.close()
Expand Down
4 changes: 2 additions & 2 deletions extractor.py
Original file line number Diff line number Diff line change
Expand Up @@ -170,8 +170,8 @@ def _extract_text(self, path: Path) -> ExtractedText:

@staticmethod
def _read_file(path: Path) -> Optional[str]:
"""Liest Datei mit UTF-8, Fallback auf cp1252 (Windows)."""
for encoding in ('utf-8', 'cp1252', 'latin-1'):
"""Liest Datei mit UTF-8 (BOM wird entfernt), Fallback auf cp1252 (Windows)."""
for encoding in ('utf-8-sig', 'cp1252', 'latin-1'):
try:
return path.read_text(encoding=encoding)
except (UnicodeDecodeError, UnicodeError):
Expand Down
Loading
Loading