M1+M2: RAG-Pipeline mit verbindlichem Grounding

agent/-Paket: Ingest (601 Layer-2-Eintraege -> 3005 Chunks, FTS5-BM25 +
Vektoren-Cache), Hybrid-Retrieval (RRF, Stand-Boost, cross_ref-Erweiterung),
Ollama-Client (embed/chat, think-Flag-Fallback, kurzes Connect-Budget),
Systemprompt mit Zitierpflicht, Post-Validierung (zitierte IDs gemaess
Retrieved-Set, 1x Regenerierung, dann Verweigerung), FastAPI (/ask, /health,
/reindex), CLI, Goldset (31 Fragen, IDs gegen kb.json verifiziert, inkl.
ATZ-Konfliktfall + 4 Verweigerungsfaelle), Eval-Suite, Test-Chat.

41 Offline-Tests gruen. Baseline BM25-only: Hit-Rate 0,871 / Recall@8 0,855 /
MRR 0,476. Hybrid-Messung, Antwortmodus-Eval und Modell-Bake-off (M3) auf
dem Host ausstaendig (Ollama aus der Zed-Sandbox nicht erreichbar).

MEMORY.md und planung.md Umsetzungsstand aktualisiert.
This commit is contained in:
2026-09-14 16:53:04 +02:00
parent bf8191b013
commit 2cba72aeb0
26 changed files with 2574 additions and 1 deletions
+108
View File
@@ -0,0 +1,108 @@
"""Schmaler Ollama-HTTP-Client (Chat + Embeddings).
Keine Tools, kein Browsing, keine Web-Hooks — der Agent hat architektonisch
keinen Weg aus der Wissensbasis hinaus (Grounding-Regel 1).
"""
from __future__ import annotations
from typing import Any
import httpx
class OllamaError(RuntimeError):
pass
class OllamaClient:
def __init__(self, base_url: str, embed_timeout_s: float = 240.0,
chat_timeout_s: float = 300.0):
self.base_url = base_url.rstrip("/")
self.embed_timeout_s = embed_timeout_s
self.chat_timeout_s = chat_timeout_s
# Per-Request-Timeouts mit kurzem Connect-Budget — ein unerreichbarer
# Server muss in Sekunden, nicht Minuten scheitern.
self.embed_timeout = httpx.Timeout(embed_timeout_s, connect=10.0)
self.chat_timeout = httpx.Timeout(chat_timeout_s, connect=10.0)
self._client = httpx.Client(timeout=self.chat_timeout)
def close(self) -> None:
self._client.close()
# -- Infrastruktur ------------------------------------------------------
def is_up(self) -> bool:
try:
r = self._client.get(f"{self.base_url}/api/tags", timeout=5.0)
return r.status_code == 200
except httpx.HTTPError:
return False
def list_models(self) -> list[str]:
r = self._client.get(f"{self.base_url}/api/tags")
r.raise_for_status()
return [m.get("model", "") for m in r.json().get("models", [])]
# -- Embeddings ---------------------------------------------------------
def embed(self, model: str, texts: list[str]) -> list[list[float]]:
"""Batch-Embeddings via /api/embed; Fallback auf Legacy /api/embeddings."""
r = self._client.post(
f"{self.base_url}/api/embed",
json={"model": model, "input": texts},
timeout=self.embed_timeout,
)
if r.status_code == 404:
out: list[list[float]] = []
for t in texts:
rr = self._client.post(
f"{self.base_url}/api/embeddings",
json={"model": model, "prompt": t},
timeout=self.embed_timeout,
)
rr.raise_for_status()
emb = rr.json().get("embedding")
if not emb:
raise OllamaError("legacy /api/embeddings returned no embedding")
out.append(emb)
return out
r.raise_for_status()
emb = r.json().get("embeddings")
if not isinstance(emb, list) or len(emb) != len(texts):
raise OllamaError(f"unexpected /api/embed response for {len(texts)} inputs")
return emb
# -- Chat ---------------------------------------------------------------
def chat(self, model: str, messages: list[dict],
temperature: float = 0.1, num_ctx: int = 16384,
num_predict: int = 1024, think: bool = False) -> str:
"""POST /api/chat, stream=False; `think`-Flag mit 404/400-Fallback."""
body: dict[str, Any] = {
"model": model,
"messages": messages,
"stream": False,
"options": {
"temperature": temperature,
"num_ctx": num_ctx,
"num_predict": num_predict,
},
}
if think is not None:
body["think"] = bool(think)
r = self._client.post(
f"{self.base_url}/api/chat", json=body, timeout=self.chat_timeout
)
if r.status_code in (400, 404) and "think" in body:
# Ältere Ollama-Versionen kennen das think-Flag nicht -> Retry ohne.
body.pop("think")
r = self._client.post(
f"{self.base_url}/api/chat", json=body, timeout=self.chat_timeout
)
r.raise_for_status()
data = r.json()
msg = data.get("message") or {}
content = msg.get("content") or ""
if not content.strip():
raise OllamaError(f"empty response from {model} (keys: {list(data.keys())})")
return content