"""Independent quality audit over a FINISHED generation run — read-only. Measures the MECE goal ("no duplicates, no gaps") with detectors that deliberately do NOT reuse the pipeline's heuristics (_canonical_key/_relation_conflict/_evidence_pack) — shared blind spots would make the audit worthless. Shared infra only: DB access, embedding.py, the agent runner (--llm sampling), atomic_write_json. CLI: python3 qa.py [--llm] (or: make qa TOPIC= [LLM=1]) Report: storage/qa//.json + console digest + diff to the previous report of the same topic. """ import asyncio import hashlib import logging import re import sys from datetime import datetime, timezone from pathlib import Path import database as db import embedding from config import JUDGE_CHUNK, STORAGE_DIR, SUB_DUP_KANDIDAT_COS from fsutil import atomic_write_json from jsonio import read_json_file as _json_file from paths import arbeit_dir from textkit import _norm_title log = logging.getLogger("creator.qa") QA_DIR = STORAGE_DIR / "qa" JACCARD_FLOOR = 0.5 # title token overlap that makes a pair suspicious EMB_FLOOR = 0.82 # casefolded title cosine (own threshold, NOT the pipeline's 0.65) SECTION_CHARS = 4000 # own paragraph splitter — independent of _text_sections COVER_MIN_TOKENS = 2 # distinctive block tokens a section must share to count as covered FREMD_MIN_TOKENS = 1 # distinctive title tokens that must appear in the corpus CONCEPT_COVER_RATIO = 0.6 # fraction of a named result's stems a block must share to count as covered LLM_SAMPLE = 12 # pairs/sections per judge call with --llm # Note 0-10, deterministisch aus den Quoten (transparent, diffbar — keine LLM-"Gefühlsnote"). # Lücken/Fremd wiegen am schwersten (fehlender/falscher Stoff). konzept_luecken (benannte Kernresultate # ohne Baustein) ist heuristisch → mittleres Gewicht. Dubletten-VERDACHT enthält bewusst Rauschen, # bildet aber das Nutzerproblem (Dopplungen) ab → leichtestes Gewicht, aber nicht null. NOTE_GEWICHTE = {"luecken": 3.0, "fremd": 2.5, "unechte_bloecke": 2.5, "konzept_luecken": 1.5, "dubletten_verdacht": 1.0, "hygiene": 0.5} # subs/artefacts only exist after board 2 — at gate time these quotas would always be 0 # and water down the inventory score, hence a separate score. # sub_dubletten counts only with --llm (confirmed pairs); the bare candidate list is # suspicion (sub_dubletten_verdacht, weightless — like dubletten_verdacht). NOTE_GEWICHTE_ARTEFAKTE = {"subs_ohne_beleg": 2.0, "verwaiste": 1.0, "sub_dubletten": 1.0} _WORD = re.compile(r"\w{3,}") _PAREN = re.compile(r"^\s*(.*?)\s*\(([^()]{2,60})\)\s*$") _STOP = {"der", "die", "das", "und", "oder", "für", "mit", "von", "des", "den", "dem", "ein", "eine", "the", "and", "for", "with", "als", "auf", "bei", "aus", "problem", "algorithmus", "algorithm", "definition", "satz", "lemma"} def _tokens(s: str) -> set[str]: return {t for t in _WORD.findall((s or "").casefold()) if t not in _STOP} def _distinctive(s: str) -> set[str]: """Tokens that can anchor a title in a corpus (stopword-free, ≥3 chars).""" return _tokens(s) def _ascii(t: str) -> str: return "".join(c for c in t if c.isascii()) def _jaccard(a: set[str], b: set[str]) -> float: return len(a & b) / len(a | b) if a | b else 0.0 def _sections(text: str, goal: int | None = None) -> list[str]: """Own paragraph-boundary splitter (NOT blocks._text_sections — independence).""" goal = goal or SECTION_CHARS out, buf = [], "" for para in re.split(r"\n\s*\n", text.strip()): para = para.strip() if not para: continue if buf and len(buf) + len(para) > goal: out.append(buf) buf = para else: buf = f"{buf}\n\n{para}" if buf else para if buf.strip(): out.append(buf) return out def _corpus_texts(topic: str) -> dict[str, str]: from blocks import source_folder # lazy: blocks pulls heavy deps folder = source_folder(topic) if not folder or not folder.is_dir(): return {} out = {} for f in sorted(folder.glob("*.txt")): try: text = f.read_text(encoding="utf-8") except OSError: continue # QUELLE:-Header (Metadata, kein Lerninhalt) strippen — sonst bildet er bei großen # Seiten einen isolierten, block-losen Abschnitt = False-Positive-Lücke. lines = text.splitlines() if lines and lines[0].startswith("QUELLE:"): text = "\n".join(lines[1:]).lstrip("\n") out[f.name] = text return out # ── Detectors ─────────────────────────────────────────────────────────────────────── def dubletten(blocks: list[dict], emb_on: bool = True) -> list[dict]: """Suspicious pairs via signal UNION: token jaccard, casefolded-title embedding cosine, paren acronym == other title. Every signal is independent of the pipeline.""" titles = [b["title"] for b in blocks] toks = [_tokens(t) for t in titles] sims = None if emb_on and titles and embedding.available(): arr = embedding.embed([t.casefold() for t in titles]) if arr is not None: sims = arr @ arr.T ops = [bool(re.search(r"[≤⪯≥⊆⊊→⇒⟹⇔←]", t)) for t in titles] out = [] for i in range(len(titles)): for j in range(i + 1, len(titles)): # relation vs. its operand ("Subset Sum" ⊂ "3-SAT ≤ Subset Sum"): by design # separate entities — token containment there is expected, not suspicious if ops[i] != ops[j] and (toks[i] <= toks[j] or toks[j] <= toks[i]): continue signals = {} jac = _jaccard(toks[i], toks[j]) if jac >= JACCARD_FLOOR: signals["jaccard"] = round(jac, 2) if sims is not None and float(sims[i][j]) >= EMB_FLOOR: signals["emb_cos"] = round(float(sims[i][j]), 2) for a, b in ((i, j), (j, i)): m = _PAREN.match(titles[a]) if m and _norm_title(titles[b]) in (_norm_title(m.group(1)), _norm_title(m.group(2))): signals["akronym"] = True if signals: out.append({"a": titles[i], "b": titles[j], "signale": signals}) return out def sub_dubletten(sub_rows: list[dict], emb_on: bool = True) -> list[dict]: """Suspicious SUB pairs, in-block AND cross-block: casefolded title cosine ≥ SUB_DUP_KANDIDAT_COS. The pipeline's own merge paths act from 0.90 upward — the measured bulk of real paraphrase duplicates sits in the band below, so everything above the floor is a candidate. The verdict falls with --llm; without it this is a suspicion list only (weightless).""" cons = [r for r in sub_rows if r["status"] == "consensus"] if len(cons) < 2 or not emb_on or not embedding.available(): return [] arr = embedding.embed([r["sub_title"].casefold() for r in cons]) if arr is None: return [] sims = arr @ arr.T out = [] for i in range(len(cons)): for j in range(i + 1, len(cons)): v = float(sims[i][j]) if v >= SUB_DUP_KANDIDAT_COS: out.append({"a": f"[{cons[i]['block']}] {cons[i]['sub_title']}", "b": f"[{cons[j]['block']}] {cons[j]['sub_title']}", "cos": round(v, 2), "cross": cons[i]["block_norm"] != cons[j]["block_norm"]}) return sorted(out, key=lambda p: -p["cos"]) def luecken(blocks: list[dict], subs_by_norm: dict[str, list[str]], corpus: dict[str, str]) -> list[dict]: """Corpus sections no block (title+description+subs tokens) sufficiently anchors. Description tokens matter at the QA GATE: board 2 has not run yet, so titles alone under-cover and inflate the quota.""" anchors: list[set[str]] = [] for b in blocks: t = _distinctive(b["title"]) | _distinctive(b.get("description") or "") for s in subs_by_norm.get(_norm_title(b["title"]), []): t |= _distinctive(s) anchors.append(t) out = [] for fname, text in corpus.items(): for k, sec in enumerate(_sections(text), 1): sec_toks = _tokens(sec) covered = any(len(a & sec_toks) >= COVER_MIN_TOKENS for a in anchors) if not covered: preview = " ".join(sec.split())[:120] out.append({"datei": fname, "abschnitt": k, "vorschau": preview}) return out # A NAMED result carries a concept name, not just a number: "Satz 7.13 (Christofides)", # "Satz 6.24: Cook-Levin". These are the core statements the token-based `luecken` misses — 108 # over-granular blocks cover every SECTION, yet the named core result may have no block at all. # Generic + domain-safe: the NUMBER is mandatory (a numbered labeled unit is a formal statement in # a structured document, any field), so prose like "ein Satz von Goethe" (no number) never matches. _NAMED_RESULT_RE = re.compile( r'\b(?:Satz|Lemma|Korollar|Theorem|Proposition|Folgerung|Definition|Algorithmus)\s+\d+(?:\.\d+)*\s*' r'(?:\(\s*([^()\n]{3,60}?)\s*\)|:\s*([^\n.;·]{3,60}?)\s*(?:[.\n;·]|$))', re.M) def _stem(t: str) -> str: """Declension-tolerant token stem: drop trailing digits, keep the 5-char prefix ('Eulerschen'/'Eulerscher' → 'euler', 'Kreise'/'Kreis' → 'kreis').""" return t.rstrip("0123456789")[:5] def _named_results(corpus: dict[str, str]) -> dict[str, set[str]]: """Named/attributed corpus results → {concept name: distinctive stems}. A bare 'Satz 7.18' (number, no name) yields nothing to match. Same catalogue vocabulary as the title strip. Umbruch-Schutz (generisch): endet ein Doppelpunkt-Fang am Zeilenumbruch und läuft die nächste Zeile klein weiter, ist es ein UMBROCHENER SATZ, kein Name — solche Fragmente („Satz 6.16: Wenn ein … in P⏎liegt …") kann kein Block je ankern (Phantom-Lücke).""" out: dict[str, set[str]] = {} for text in corpus.values(): for m in _NAMED_RESULT_RE.finditer(text): name = (m.group(1) or m.group(2) or "").strip() if m.group(2) and m.end() > 0 and text[m.end() - 1] == "\n": rest = text[m.end():].lstrip(" \t") if rest[:1].islower(): continue # Satzfragment: Fortsetzung klein auf der nächsten Zeile toks = _distinctive(name) if len(name) >= 3 and toks: out.setdefault(name, set()).update(_stem(t) for t in toks) return out def konzept_luecken(blocks: list[dict], named: dict[str, set[str]]) -> list[str]: """Named corpus results that NO block covers — concept gaps the token-based `luecken` cannot see. Covered = a block whose title+description share ≥ CONCEPT_COVER_RATIO of the result's distinctive stems (declension-tolerant). Errs toward 'covered' so the heuristic never invents a false gap.""" anchors = [{_stem(t) for t in _distinctive(b["title"]) | _distinctive(b.get("description") or "")} for b in blocks] out = [] for name, stems in named.items(): if not any(len(stems & a) >= CONCEPT_COVER_RATIO * len(stems) for a in anchors): out.append(name) return sorted(out) def fremd(blocks: list[dict], corpus: dict[str, str]) -> list[str]: """Blocks whose distinctive title tokens never appear in the corpus (scope creep). Token/stem match, NOT raw substring — 'bergang' ⊂ 'Übergang' had whitewashed the garbage title 'αÜbergang'. The ASCII form only bridges symbol variants (Δ/∆).""" ctoks = set(_WORD.findall("\n".join(corpus.values()).casefold())) def _hit(t: str) -> bool: forms = {t} | ({a} if len(a := _ascii(t)) >= 3 else set()) # digit-suffix fallback: '∆TSP1' → 'tsp1' misses the corpus token 'tsp' ('∆' is no \w) forms |= {f2 for f in list(forms) if len(f2 := f.rstrip("0123456789")) >= 3} return any(ct == f or ct.startswith(f) for f in forms for ct in ctoks) out = [] for b in blocks: dist = _distinctive(b["title"]) if dist and sum(1 for t in dist if _hit(t)) < FREMD_MIN_TOKENS: out.append(b["title"]) return out def beleg(blocks: list[dict], sub_rows: list[dict]) -> dict: ohne_quelle = [b["title"] for b in blocks if not b.get("sources")] ohne_mention = [f"{r['block']} · {r['sub_title']}" for r in sub_rows if r["status"] != "variant" and not r["mentions"]] return {"bloecke_ohne_quelle": ohne_quelle, "subs_ohne_beleg": ohne_mention} def hygiene(blocks: list[dict]) -> list[dict]: out = [] for b in blocks: t = b["title"] probleme = [] if "**" in t or "`" in t: probleme.append("markdown") if re.search(r"\(\d+\)\s*$", t): probleme.append("kollisions-suffix") if not (b.get("description") or "").strip(): probleme.append("leere-beschreibung") if probleme: out.append({"titel": t, "probleme": probleme}) return out def _zaehlbare_luecken(lk: list[dict], llm: bool) -> list[dict]: """With --llm only non-refuted gaps count ('?' = unjudged stays, conservative) — refuted ones dragged the note although the judge cleared them (aak: 5 of 8, weight 3.0). Persistierte Freisprüche (2:1 im Repair) zählen nie.""" offen = [x for x in lk if not x.get("freispruch")] return [x for x in offen if x.get("llm") != "nein"] if llm else offen def note(quoten: dict, gewichte: dict = NOTE_GEWICHTE) -> float: """10 = alle gewichteten Quoten 0. Gewicht = Punktabzug bei 100 % Quote (keine Normierung, sonst staucht die Gewichtssumme die Skala nach oben). Ungemessene Quoten zählen nicht — unechte_bloecke existiert nur mit --llm; dubletten_verdacht ist Verdachtsliste, kein Urteil.""" da = {k: w for k, w in gewichte.items() if k in quoten} schaden = sum(w * min(float(quoten[k]), 1.0) for k, w in da.items()) return round(max(0.0, 10.0 * (1 - schaden)), 1) def artefakte(sub_rows: list[dict], art_rows: list[dict], fragen: list[dict]) -> dict: """Coverage + orphans of the learning artefacts. Nenner = consensus-Subs (verworfene zählen nicht als abzudeckendes Material). Waise = Ziel weder lebend (consensus/variant) noch eindeutig als Kurztitel-Präfix von „kurztitel: beschreibung" auflösbar.""" if not art_rows and not fragen: return {"status": "nicht generiert"} cons = {(r["block_norm"], r["sub_norm"]) for r in sub_rows if r["status"] == "consensus"} lebt = {(r["block_norm"], r["sub_norm"]) for r in sub_rows if r["status"] != "discarded"} def _ziel(bn: str, sn: str): if (bn, sn) in lebt: return (bn, sn) treffer = [k for k in lebt if k[0] == bn and k[1].startswith(sn + ":")] if len(treffer) == 1: return treffer[0] # mehrere Treffer = meist ein consensus-Sub plus seine gefalteten Varianten haupt = [k for k in treffer if k in cons] return haupt[0] if len(haupt) == 1 else None deck: dict[str, set] = {} verwaist = [] for typ, bn, sn in ([(r["type"], r["block_norm"], r["sub_norm"]) for r in art_rows] + [("frage", r["block_norm"], r["sub_norm"]) for r in fragen]): z = _ziel(bn, sn) if z is None: verwaist.append(f"{typ}: {bn} · {sn}") else: deck.setdefault(typ, set()).add(z) n = max(len(cons), 1) return {"status": "ok", "frage_abdeckung": round(len(deck.get("frage", set()) & cons) / n, 3), "flashcard_abdeckung": round(len(deck.get("flashcard", set()) & cons) / n, 3), "beispiel_abdeckung": round(len(deck.get("example", set()) & cons) / n, 3), "verwaiste": sorted(verwaist)} # ── LLM sampling (optional, own prompts under templates/QA/) ──────────────────────── def _qa_prompt(name: str, **kwargs) -> str: """Own template dir (templates/QA/) — deliberately separate from the pipeline prompts.""" from config import TEMPLATES_DIR return (TEMPLATES_DIR / "QA" / f"{name}.md").read_text(encoding="utf-8").format(**kwargs) async def judge_wave(template: str, topic: str, key: str, slot: str, items: list[str], *, chunk: int = JUDGE_CHUNK, prefix: str = "qa", label: str = "QA", extra: str = "") -> dict[int, str]: """Gechunkte Ja/Nein-Judge-Welle über ALLE Items, Chunks parallel (die Semaphoren in agents.py begrenzen); Ergebnis mit globalen 1-basierten Indizes. Fail-open pro Chunk (Items bleiben ohne Urteil), aber nie stumm. Ersetzt die drei strukturgleichen Handkopien in repair/qa/guide_qa. `extra` füllt den {extra}-Slot des Templates (z. B. Inventar-Kontext für das Lücken-Urteil im Repair).""" from agents import run_agent from pipeline import _timeout, _yesno_schema from jsonio import parse_json_text async def _chunk(lo: int) -> dict[int, str]: teil = items[lo:lo + chunk] listing = "\n\n".join(f"{k}. {it}" for k, it in enumerate(teil, 1)) try: rc, out, _err = await run_agent( f"{prefix}-{topic}-{key}-{lo}", _qa_prompt(template, topic=topic, extra=extra, **{slot: listing}), _timeout("qa_judge"), role="judge", capabilities="none", scope=topic, label=f"{label} {key}") except Exception: log.exception("[%s] %s-Judge %s+%d fehlgeschlagen — Items ohne Urteil", topic, label, key, lo) return {} if rc != 0: log.warning("[%s] %s-Judge %s+%d fehlgeschlagen (rc=%s) — %d Items ohne Urteil", topic, label, key, lo, rc, len(teil)) return {} return _yesno_schema(parse_json_text(out)) or {} offsets = range(0, len(items), chunk) results = await asyncio.gather(*[_chunk(lo) for lo in offsets]) return {lo + k: urteil for lo, v in zip(offsets, results) for k, urteil in v.items()} # ── Report ────────────────────────────────────────────────────────────────────────── def report_paths(topic: str, guide: bool = False) -> list[Path]: """QA-Reports eines Topics, mtime-aufsteigend (Run-ID- und Timestamp-Namen sortieren lexikographisch nicht). guide=True → die separate guide-*-Serie (guide_qa.py). freispruch.json teilt den Ordner, ist aber kein Report — immer außen vor.""" tdir = QA_DIR / topic if not tdir.is_dir(): return [] return sorted((p for p in tdir.glob("*.json") if p.name.startswith("guide-") == guide and p.name != "freispruch.json"), key=lambda p: p.stat().st_mtime) _latest_cache: dict[tuple[str, bool], tuple[float, dict]] = {} def latest_report(topic: str, guide: bool = False) -> dict | None: """Jüngster Report als geparstes dict, mtime-gecacht — die Board-Snapshots lesen das im 1,2-s-Frontend-Takt, ein JSON-Read je Poll war unnötiges Datei-I/O. glob+stat bleiben (billig), der Read passiert nur bei geänderter mtime.""" reports = report_paths(topic, guide) if not reports: return None p = reports[-1] mtime = p.stat().st_mtime key = (topic, guide) cached = _latest_cache.get(key) if cached is None or cached[0] != mtime: _latest_cache[key] = (mtime, _json_file(p) or {}) return _latest_cache[key][1] def freispruch_pfad(topic: str) -> Path: return QA_DIR / topic / "freispruch.json" def _paar_key(a: str, b: str) -> str: return "||".join(sorted((_norm_title(a), _norm_title(b)))) def luecken_key(item: dict) -> str: """Stabiler Freispruch-Schlüssel einer Lücken-Fundstelle: Datei + Vorschau-Hash. Abschnittsnummern verschieben sich bei Korpusänderung — dann veralten die Keys und werden inert (gewollt: neue Fundstellen brauchen ein neues Urteil).""" v = " ".join(str(item.get("vorschau", "")).casefold().split()) return f"{item.get('datei', '')}||{hashlib.sha1(v.encode()).hexdigest()[:16]}" def offene_luecken(report: dict) -> tuple[list[dict], list[str]]: """Zählbare offene Lücken eines Reports: (luecken-Items ohne Freispruch/LLM-nein, konzept_luecken-Namen). Grundlage für Lücken-Research und Lücken-Urteil im Repair.""" lk = [x for x in report.get("luecken", []) if not x.get("freispruch") and x.get("llm") != "nein"] return lk, list(report.get("konzept_luecken", [])) def lade_freispruch(topic: str) -> dict[str, list[str]]: """Persistierte 2:1-Freisprüche des Repair-Stichentscheids (repair._mit_stichentscheid): mehrheitlich als „behalten" geurteilte Befunde zählen nicht mehr in die Note — sonst pendelte sie dauerhaft unter 10 ohne Fix-Pfad (gemessen: kanban-smoke 9.4, aak 9.2). Die Detektoren bleiben unverändert; ein Freispruch ist ein persistiertes Urteil, kein Detektor-Tuning. Freigesprochene bleiben im Report sichtbar.""" d = _json_file(freispruch_pfad(topic)) return d if isinstance(d, dict) else {} async def qa_report(topic: str, llm: bool = False) -> dict | None: cards = await db.kanban_cards(topic, board="inventory", stage="done_block") if not cards: print(f"Keine fertigen Blöcke für '{topic}' — Tippfehler im Namen oder Lauf nicht durch?") return None blocks = [{"title": c["payload"].get("title", ""), "description": c["payload"].get("description", ""), "sources": c["payload"].get("sources") or []} for c in cards] sub_rows = [dict(r) for bn in {_norm_title(b["title"]) for b in blocks} for r in await db.list_subblocks(topic, bn)] subs_by_norm: dict[str, list[str]] = {} for r in sub_rows: if r["status"] != "variant": subs_by_norm.setdefault(r["block_norm"], []).append(r["sub_title"]) corpus = _corpus_texts(topic) d = dubletten(blocks) sd = sub_dubletten(sub_rows) lk = luecken(blocks, subs_by_norm, corpus) if corpus else [] fr = fremd(blocks, corpus) if corpus else [] named = _named_results(corpus) if corpus else {} kl = konzept_luecken(blocks, named) if corpus else [] bl = beleg(blocks, sub_rows) hy = hygiene(blocks) n_sections = sum(len(_sections(t)) for t in corpus.values()) or 1 frei = lade_freispruch(topic) frei_fremd = set(frei.get("fremd") or []) fremd_frei = [t for t in fr if _norm_title(t) in frei_fremd] fr = [t for t in fr if _norm_title(t) not in frei_fremd] # Persistierte 2:1-Freisprüche: markiert bleibt sichtbar (diffbar), zählt aber nicht mehr # in Quote und Judge-Welle — sonst pendelte der Verdachts-Floor ewig unter 10. frei_dub = set(frei.get("dubletten") or []) for p in d: if _paar_key(p["a"], p["b"]) in frei_dub: p["freispruch"] = True d_offen = [p for p in d if not p.get("freispruch")] frei_lk = set(frei.get("luecken") or []) for x in lk: if luecken_key(x) in frei_lk: x["freispruch"] = True lk_offen = [x for x in lk if not x.get("freispruch")] frei_kl = set(frei.get("konzept_luecken") or []) kl_frei = [nm for nm in kl if _norm_title(nm) in frei_kl] kl = [nm for nm in kl if _norm_title(nm) not in frei_kl] if llm and d_offen: v = await judge_wave("QA-Dubletten", topic, "dubletten", "pairs", [f"A: {p['a']}\nB: {p['b']}" for p in d_offen[:LLM_SAMPLE]]) for k, p in enumerate(d_offen[:LLM_SAMPLE], 1): p["llm"] = v.get(k, "?") if llm and lk_offen: v = await judge_wave("QA-Luecken", topic, "luecken", "sections", [f"[{x['datei']} #{x['abschnitt']}] {x['vorschau']}" for x in lk_offen[:LLM_SAMPLE]]) for k, x in enumerate(lk_offen[:LLM_SAMPLE], 1): x["llm"] = v.get(k, "?") if llm and sd: # full coverage in chunks — a sampled quota would mislead the note v = await judge_wave("QA-Sub-Dubletten", topic, "sub-dubletten", "pairs", [f"A: {p['a']}\nB: {p['b']}" for p in sd]) for k, p in enumerate(sd, 1): p["llm"] = v.get(k, "?") frei_sub = set(frei.get("sub_dubletten") or []) for p in sd: if p.get("llm") == "ja" and _paar_key(p["a"], p["b"]) in frei_sub: p["freispruch"] = True # 2:1-Urteil „behalten" — sichtbar, aber notenfrei unecht: list[str] | None = None if llm and blocks: v = await judge_wave("QA-Bausteine", topic, "bausteine", "blocks", [f"{b['title']} — {b['description'] or '(ohne Beschreibung)'}" for b in blocks], chunk=80) verdacht = [b for k, b in enumerate(blocks, 1) if v.get(k) == "nein"] # Bestätiger-Pass nur über die Geflaggten: der Einzel-Judge flaggte pro Lauf ANDERE # Blöcke (gemessen aak: Note pendelte 9.3↔10.0 bei identischem Bestand) — nur # doppelt-„nein" zählt; Repair hat als dritte Sicherung die eigene Zweitmeinung unecht = [] if verdacht: v2 = await judge_wave("QA-Bausteine", topic, "bausteine-b2", "blocks", [f"{b['title']} — {b['description'] or '(ohne Beschreibung)'}" for b in verdacht]) unecht = [b["title"] for k, b in enumerate(verdacht, 1) if v2.get(k) == "nein"] frei_unecht = set(frei.get("unecht") or []) unecht = [t for t in unecht if _norm_title(t) not in frei_unecht] art_rows = [dict(r) for r in await db.get_sub_artefakte(topic)] fragen = [dict(r) for r in await db.list_question_pattern(topic)] art = artefakte(sub_rows, art_rows, fragen) quoten_art: dict[str, float] = {} if sub_rows: quoten_art["subs_ohne_beleg"] = round(len(bl["subs_ohne_beleg"]) / len(sub_rows), 3) if art.get("status") == "ok": quoten_art["verwaiste"] = round(len(art["verwaiste"]) / max(len(art_rows) + len(fragen), 1), 3) n_cons = sum(1 for r in sub_rows if r["status"] == "consensus") if n_cons: quoten_art["sub_dubletten_verdacht"] = round(len(sd) / n_cons, 3) if llm: # confirmed pairs only — the bare candidate list is suspicion, not damage; # freigesprochene (2:1 „behalten") zählen nicht mehr quoten_art["sub_dubletten"] = round( sum(1 for p in sd if p.get("llm") == "ja" and not p.get("freispruch")) / n_cons, 3) summary = _json_file(arbeit_dir(topic) / "lauf-summary.json") or {} report = { "topic": topic, "erstellt": datetime.now(timezone.utc).isoformat(), "run_id": summary.get("run_id", ""), "bloecke": len(blocks), "quoten": { "dubletten_verdacht": round(len(d_offen) / max(len(blocks), 1), 3), "luecken": round(len(_zaehlbare_luecken(lk, llm)) / n_sections, 3), "fremd": round(len(fr) / max(len(blocks), 1), 3), **({"konzept_luecken": round(len(kl) / max(len(named), 1), 3)} if corpus else {}), "hygiene": round(len(hy) / max(len(blocks), 1), 3), **({"unechte_bloecke": round(len(unecht) / max(len(blocks), 1), 3)} if unecht is not None else {}), }, "quoten_artefakte": quoten_art, **({"unecht": unecht} if unecht is not None else {}), **({"fremd_freigesprochen": fremd_frei} if fremd_frei else {}), **({"konzept_luecken_freigesprochen": kl_frei} if kl_frei else {}), "dubletten": d, "sub_dubletten": sd, "luecken": lk, "konzept_luecken": kl, "fremd": fr, "beleg": bl, "hygiene": hy, "artefakte": art, "lauf": summary, } report["note"] = note(report["quoten"]) # None statt 10.0, solange Board 2 nichts geliefert hat — nichts gemessen ist keine Bestnote report["note_artefakte"] = note(quoten_art, NOTE_GEWICHTE_ARTEFAKTE) if quoten_art else None report["note_gewichte"] = {"inventar": NOTE_GEWICHTE, "artefakte": NOTE_GEWICHTE_ARTEFAKTE} return report def _diff(prev: dict | None, cur: dict) -> dict: if not prev: return {} # ältere Reports führten die Artefakt-Quoten noch unter "quoten" alt = {**prev.get("quoten", {}), **prev.get("quoten_artefakte", {})} neu = {**cur["quoten"], **cur.get("quoten_artefakte", {})} return {k: round(v - alt.get(k, 0), 3) for k, v in neu.items()} def _write_report(report: dict) -> Path: tdir = QA_DIR / report["topic"] tdir.mkdir(parents=True, exist_ok=True) older = report_paths(report["topic"]) prev = _json_file(older[-1]) if older else None report["diff_zum_vorlauf"] = _diff(prev, report) name = report["run_id"] or datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S") path = tdir / f"{name}.json" atomic_write_json(path, report, indent=1) return path async def write_report(report: dict) -> Path: """_write_report + kompaktes kind='qa'-Event. Die Report-JSONs liegen nur auf der Lauf-Maschine (storage/qa/) — ein DB-Pull reichte nicht, um Note/Quoten eines Runs zu rekonstruieren (Analyse 20260704-1452-b223). Nur die Kennzahlen, kein Volltext; run_id stempelt add_event aus der Registry (gesetzt im Lauf, leer bei manueller QA).""" path = await asyncio.to_thread(_write_report, report) try: # Event ist Komfort — ein DB-Fehler darf den Report nicht kosten (fail-open) await db.add_event(report["topic"], "qa", key=path.stem, meta={ "note": report["note"], "note_artefakte": report.get("note_artefakte"), "quoten": report["quoten"], "quoten_artefakte": report.get("quoten_artefakte", {})}) except Exception: pass return path def _digest(report: dict, path: Path): na = report.get("note_artefakte") print(f"QA {report['topic']} — {report['bloecke']} Blöcke (run {report['run_id'] or '—'})" f" — Inventar {report['note']}/10 · Artefakte {f'{na}/10' if na is not None else '—'}") for k, v in {**report["quoten"], **report.get("quoten_artefakte", {})}.items(): delta = report.get("diff_zum_vorlauf", {}).get(k) d = f" ({'+' if delta > 0 else ''}{delta})" if delta else "" print(f" {k:20} {v:6.1%}{d}") for p in report["dubletten"][:8]: print(f" DUBLETTE? {p['a']} <-> {p['b']} {p['signale']}{' LLM:' + p['llm'] if 'llm' in p else ''}") for p in report.get("sub_dubletten", [])[:8]: print(f" SUB-DUP? {p['a']} <-> {p['b']} cos={p['cos']}{' LLM:' + p['llm'] if 'llm' in p else ''}") for t in report["fremd"][:8]: print(f" FREMD? {t}") for t in report.get("unecht", [])[:8]: print(f" UNECHT {t}") art = report["artefakte"] if art.get("status") == "ok": print(f" Artefakte: Frage {art['frage_abdeckung']:.0%} · Flashcard {art['flashcard_abdeckung']:.0%}" f" · Beispiel {art['beispiel_abdeckung']:.0%} · verwaist {len(art['verwaiste'])}") else: print(" Artefakte: nicht generiert (Board 2 nicht gelaufen)") print(f"Report: {path}") async def main(topic: str, llm: bool): await db.init_db() try: report = await qa_report(topic, llm=llm) if report is None: sys.exit(1) _digest(report, await write_report(report)) finally: await db.close_db() if __name__ == "__main__": args = [a for a in sys.argv[1:] if not a.startswith("--")] if not args: print("Nutzung: python3 qa.py [--llm]") sys.exit(1) asyncio.run(main(args[0], "--llm" in sys.argv))