"""Index-Bau: Layer-2-Einträge -> SQLite (FTS5-BM25 + Dense-Vektoren). Chunking: H2-Sektion je Eintrag (Parent-Child: Retrieval auf Sektion, Kontext = Sektion + Metadatenkopf). Die Vektoren-Tabelle ist ein Cache (Content-Hash) und überlebt Rebuilds — ein Reindex bettet nur Neues ein. """ from __future__ import annotations import hashlib import json import sqlite3 import time from dataclasses import asdict, dataclass from datetime import datetime, timezone from pathlib import Path import numpy as np from .config import Config from .kb import KBEntry, Section, load_kb from .normalize import normalize_text SCHEMA_VERSION = 1 EMBED_BATCH = 32 SCHEMA = """ CREATE TABLE IF NOT EXISTS meta (key TEXT PRIMARY KEY, value TEXT); CREATE TABLE IF NOT EXISTS chunks ( chunk_id INTEGER PRIMARY KEY, entry_id TEXT NOT NULL, section TEXT NOT NULL, text TEXT NOT NULL, norm TEXT NOT NULL, content_hash TEXT NOT NULL, title TEXT, work TEXT, chapter TEXT, topic TEXT, stand TEXT, batch INTEGER, tags TEXT, legal_bases TEXT, cross_refs TEXT, source_pdf TEXT, source_text TEXT ); CREATE INDEX IF NOT EXISTS idx_chunks_entry ON chunks(entry_id); CREATE VIRTUAL TABLE IF NOT EXISTS chunks_fts USING fts5(norm); CREATE TABLE IF NOT EXISTS vectors ( content_hash TEXT NOT NULL, model TEXT NOT NULL, dim INTEGER NOT NULL, vec BLOB NOT NULL, PRIMARY KEY (content_hash, model) ); """ @dataclass class IndexStats: n_entries: int = 0 n_chunks: int = 0 n_embedded: int = 0 embed_error: str | None = None duration_s: float = 0.0 kb_dir: str = "" db_path: str = "" schema_version: int = SCHEMA_VERSION def as_dict(self) -> dict: return asdict(self) def embed_text(title: str, section_title: str, text: str) -> str: """Einheitlicher Embedding-Input (Title + Abschnitt + Body).""" return f"{title}\n{section_title}\n\n{text}" def content_hash_for(title: str, section_title: str, text: str) -> str: return hashlib.sha256( embed_text(title, section_title, text).encode("utf-8") ).hexdigest() def norm_text_for(entry: KBEntry, section: Section) -> str: """FTS-Text: Titel + Abschnitt + Tags + Rechtsgrundlagen + Kapitel + Body.""" parts = [ entry.title, section.title, " ".join(entry.tags), " ".join(entry.legal_bases), entry.chapter, section.text, ] return normalize_text("\n".join(p for p in parts if p)) def _insert_chunk(con: sqlite3.Connection, entry: KBEntry, section: Section) -> int: h = content_hash_for(entry.title, section.title, section.text) cur = con.execute( """INSERT INTO chunks ( entry_id, section, text, norm, content_hash, title, work, chapter, topic, stand, batch, tags, legal_bases, cross_refs, source_pdf, source_text ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", ( entry.id, section.title, section.text, norm_text_for(entry, section), h, entry.title, entry.work, entry.chapter, entry.topic, entry.stand, entry.batch, json.dumps(entry.tags, ensure_ascii=False), json.dumps(entry.legal_bases, ensure_ascii=False), json.dumps(entry.cross_refs, ensure_ascii=False), entry.source.get("pdf", ""), entry.source.get("text", ""), ), ) chunk_id = cur.lastrowid con.execute( "INSERT INTO chunks_fts(rowid, norm) VALUES (?, ?)", (chunk_id, norm_text_for(entry, section)), ) return chunk_id def _embed_missing(cfg: Config, con: sqlite3.Connection, client) -> tuple[int, str | None]: """Bettet alle Chunk-Hashes ein, die für cfg.embed_model fehlen.""" rows = con.execute( "SELECT DISTINCT content_hash, title, section, text FROM chunks" ).fetchall() unique: dict[str, str] = {} for h, title, section, text in rows: if h not in unique: unique[h] = embed_text(title, section, text) have = { r[0] for r in con.execute( "SELECT content_hash FROM vectors WHERE model = ?", (cfg.embed_model,), ).fetchall() } todo = [h for h in unique if h not in have] if not todo: return 0, None n = 0 for i in range(0, len(todo), EMBED_BATCH): batch = todo[i:i + EMBED_BATCH] texts = [unique[h] for h in batch] try: embs = client.embed(cfg.embed_model, texts) except Exception as e: # httpx/Ollama-Fehler -> BM25-only weiterlaufen return n, f"embedding failed at batch {i // EMBED_BATCH + 1}: {e}" for h, vec in zip(batch, embs): arr = np.asarray(vec, dtype=np.float32) con.execute( "INSERT OR REPLACE INTO vectors(content_hash, model, dim, vec) " "VALUES (?, ?, ?, ?)", (h, cfg.embed_model, int(arr.shape[0]), arr.tobytes()), ) n += 1 return n, None def build_index(cfg: Config, client=None) -> IndexStats: """Vollständiger Rebuild von chunks/FTS; Vektoren-Cache bleibt erhalten.""" t0 = time.perf_counter() entries = load_kb(cfg.kb_dir, verify_registry=True) db_path = Path(cfg.db_path) db_path.parent.mkdir(parents=True, exist_ok=True) stats = IndexStats(kb_dir=str(cfg.kb_dir), db_path=str(db_path)) con = sqlite3.connect(db_path) try: con.executescript(SCHEMA) con.execute("DELETE FROM chunks") con.execute("DELETE FROM chunks_fts") con.execute("DELETE FROM meta") for entry in entries: for section in entry.sections: _insert_chunk(con, entry, section) con.commit() stats.n_entries = len(entries) stats.n_chunks = con.execute("SELECT COUNT(*) FROM chunks").fetchone()[0] if not cfg.embed_off: if client is None: from .ollama_client import OllamaClient client = OllamaClient( cfg.ollama_url, embed_timeout_s=cfg.embed_timeout_s, chat_timeout_s=cfg.chat_timeout_s, ) try: stats.n_embedded, stats.embed_error = _embed_missing(cfg, con, client) except Exception as e: stats.embed_error = f"{type(e).__name__}: {e}" con.executemany( "INSERT OR REPLACE INTO meta(key, value) VALUES (?, ?)", [ ("schema_version", str(SCHEMA_VERSION)), ("built_at", datetime.now(timezone.utc).isoformat()), ("kb_dir", str(cfg.kb_dir)), ("embed_model", "" if cfg.embed_off else cfg.embed_model), ("n_entries", str(stats.n_entries)), ("n_chunks", str(stats.n_chunks)), ], ) con.commit() finally: con.close() stats.duration_s = round(time.perf_counter() - t0, 2) return stats