update
This commit is contained in:
14
Makefile
14
Makefile
@@ -1,4 +1,4 @@
|
|||||||
.PHONY: install dev prod stop logs remove auth sync sync-projects sync-all sync-all-reverse projects searxng ollama
|
.PHONY: install dev prod stop logs remove auth sync sync-projects sync-all sync-all-reverse projects searxng ollama qa
|
||||||
|
|
||||||
COMPOSE = docker compose
|
COMPOSE = docker compose
|
||||||
|
|
||||||
@@ -101,4 +101,16 @@ sync-all-reverse: stop
|
|||||||
ssh root@178.104.67.87 'cd /var/www/creator && docker compose up -d --build'
|
ssh root@178.104.67.87 'cd /var/www/creator && docker compose up -d --build'
|
||||||
@echo "Reverse-Sync abgeschlossen — Remote läuft wieder."
|
@echo "Reverse-Sync abgeschlossen — Remote läuft wieder."
|
||||||
|
|
||||||
|
# QA-Report über einen abgeschlossenen Lauf (read-only): make qa TOPIC=aak
|
||||||
|
qa:
|
||||||
|
@[ -n "$(TOPIC)" ] || { echo "Nutzung: make qa TOPIC=<thema> [LLM=1]"; exit 1; }
|
||||||
|
@set -a; [ -f .env ] && . ./.env; set +a; \
|
||||||
|
cd backend && python3 qa.py "$(TOPIC)" $(if $(LLM),--llm,)
|
||||||
|
|
||||||
|
# Guide-QA über einen gebauten Guide (read-only): make qa-guide TOPIC=Markdown [LLM=1]
|
||||||
|
qa-guide:
|
||||||
|
@[ -n "$(TOPIC)" ] || { echo "Nutzung: make qa-guide TOPIC=<thema> [LLM=1]"; exit 1; }
|
||||||
|
@set -a; [ -f .env ] && . ./.env; set +a; \
|
||||||
|
cd backend && python3 guide_qa.py "$(TOPIC)" $(if $(LLM),--llm,)
|
||||||
|
|
||||||
projects: sync-projects
|
projects: sync-projects
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ the full list.
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
import hashlib
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import math
|
import math
|
||||||
@@ -35,7 +36,7 @@ from pipeline import (
|
|||||||
)
|
)
|
||||||
from textkit import (
|
from textkit import (
|
||||||
_unique_title, _load_blocks, _norm_title, _parse_selection, _parse_subblocks, _title,
|
_unique_title, _load_blocks, _norm_title, _parse_selection, _parse_subblocks, _title,
|
||||||
_resolve_title, _title_index,
|
_resolve_title, _title_index, clean_title,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Chunk the subblocks (web search per block): 1 agent per ~10 blocks, capped.
|
# Chunk the subblocks (web search per block): 1 agent per ~10 blocks, capped.
|
||||||
@@ -55,8 +56,9 @@ RESEARCH_SECTION_CHARS = 12000
|
|||||||
SUBBLOCK_CAP = 900 # subblock find loop per chunk (15 min)
|
SUBBLOCK_CAP = 900 # subblock find loop per chunk (15 min)
|
||||||
SUBBLOCK_MIN = 5 # below this consensus count a block gets focused catch-up rounds
|
SUBBLOCK_MIN = 5 # below this consensus count a block gets focused catch-up rounds
|
||||||
SUBBLOCK_EXTRA_ROUNDS = 2 # max catch-up rounds (saturation stop still applies — thin stays thin)
|
SUBBLOCK_EXTRA_ROUNDS = 2 # max catch-up rounds (saturation stop still applies — thin stays thin)
|
||||||
SUBBLOCK_MAX_ROUNDS = 5 # hard round cap: measured runs hit 5–9 rounds purely on paraphrases
|
SUBBLOCK_MAX_ROUNDS = 3 # hard round cap: measured, rounds 4–5 burned 29 % of the finder agents
|
||||||
# before the variant-robust `new` count converges — never search longer
|
# for ~zero consensus gain (fringe ideas never saturate) — thin blocks
|
||||||
|
# are caught by the SUBBLOCK_MIN catch-up plus the gap follow-up round
|
||||||
CONSOLIDATION_CHUNK = 600 # up to here ONE global judge (dedups everything); above that chunked + merge pass — fallback path only
|
CONSOLIDATION_CHUNK = 600 # up to here ONE global judge (dedups everything); above that chunked + merge pass — fallback path only
|
||||||
DEDUP_PAIR_FLOOR = 0.6 # min cosine for a candidate pair (complete-link aggregates → no chaining)
|
DEDUP_PAIR_FLOOR = 0.6 # min cosine for a candidate pair (complete-link aggregates → no chaining)
|
||||||
DEDUP_PAIRS_CHUNK = 40 # pairs per judge package (pairwise verification instead of a block mixer)
|
DEDUP_PAIRS_CHUNK = 40 # pairs per judge package (pairwise verification instead of a block mixer)
|
||||||
@@ -65,7 +67,8 @@ DEDUP_GLOBAL_FLOOR = 0.65 # global post-naming dedup stage: candidate floor ab
|
|||||||
# same-domain noise band, below the sibling zone (~0.85) — the judge decides there
|
# same-domain noise band, below the sibling zone (~0.85) — the judge decides there
|
||||||
FILTER_CHUNK = 35 # blocks to assess per judge in the degrade pass (full list as context)
|
FILTER_CHUNK = 35 # blocks to assess per judge in the degrade pass (full list as context)
|
||||||
# Balance question-pattern chunks by sub load via LPT (makespan), not by block count.
|
# Balance question-pattern chunks by sub load via LPT (makespan), not by block count.
|
||||||
QUESTION_CHUNK_SUBS = 50 # target sum of relevant subs per chunk
|
QUESTION_CHUNK_SUBS = 25 # target sum of relevant subs per chunk — at 50 the generator
|
||||||
|
# skipped so many subs that 60 % of all question calls were catch-up
|
||||||
QUESTION_MAX_ROUNDS = 3 # catch-up rounds for subs without a pattern (the LLM omits ~18 % per chunk)
|
QUESTION_MAX_ROUNDS = 3 # catch-up rounds for subs without a pattern (the LLM omits ~18 % per chunk)
|
||||||
FACTS_CHUNK_SUBS = 10 # facts extraction: small chunks — the 4 phases (find/erg/check/fix) are
|
FACTS_CHUNK_SUBS = 10 # facts extraction: small chunks — the 4 phases (find/erg/check/fix) are
|
||||||
# serial PER CHUNK, so chunk count = parallelism; the makespan tail of a
|
# serial PER CHUNK, so chunk count = parallelism; the makespan tail of a
|
||||||
@@ -695,13 +698,17 @@ def _lpt_chunks(weights: list[int], target: int) -> list[list[int]]:
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
_NEG_TOKENS = {"nicht", "kein", "keine", "keinen", "keiner", "ohne", "nie"}
|
# lemmatized: 'kein Syntaxfehler' vs 'keine Syntax-Fehlermeldung' are the SAME statement —
|
||||||
|
# unlemmatized token sets ({kein} ≠ {keine}) blocked that fold at cos 0.974 (measured).
|
||||||
|
_NEG_LEMMA = {"nicht": "nicht", "ohne": "ohne", "nie": "nie", "niemals": "nie",
|
||||||
|
"kein": "kein", "keine": "kein", "keinen": "kein", "keiner": "kein",
|
||||||
|
"keinem": "kein", "keines": "kein"}
|
||||||
|
|
||||||
|
|
||||||
def _neg_set(title: str) -> frozenset:
|
def _neg_set(title: str) -> frozenset:
|
||||||
"""Negation tokens of a title — antonym statements measure cos 0.91–0.95 (above any usable
|
"""Lemmatized negation tokens of a title — antonym statements measure cos 0.91–0.95
|
||||||
variant threshold), so equal negation sets are a hard merge precondition."""
|
(above any usable variant threshold), so equal negation sets are a hard merge precondition."""
|
||||||
return frozenset(t for t in re.findall(r"\w+", _norm_title(title)) if t in _NEG_TOKENS)
|
return frozenset(l for t in re.findall(r"\w+", _norm_title(title)) if (l := _NEG_LEMMA.get(t)))
|
||||||
|
|
||||||
|
|
||||||
def _sub_tokens(title: str) -> set:
|
def _sub_tokens(title: str) -> set:
|
||||||
@@ -1097,7 +1104,6 @@ async def _subblocks_block(ctx: GenContext, set_p, files: dict, entries: dict, i
|
|||||||
_report_p(set_p, topic, "Subblocks clarify"))
|
_report_p(set_p, topic, "Subblocks clarify"))
|
||||||
if is_cancelled():
|
if is_cancelled():
|
||||||
return None
|
return None
|
||||||
await _dedup_subblocks(topic, raw) # near-dup filter per block (deterministic, no LLM)
|
|
||||||
|
|
||||||
# Seed guarantee (single-block kanban calls): every demoted-fragment seed must reach the
|
# Seed guarantee (single-block kanban calls): every demoted-fragment seed must reach the
|
||||||
# facts evidence gate — covered by a consensus sub, promoted from a single find, or
|
# facts evidence gate — covered by a consensus sub, promoted from a single find, or
|
||||||
@@ -1130,6 +1136,9 @@ async def _subblocks_block(ctx: GenContext, set_p, files: dict, entries: dict, i
|
|||||||
raw.setdefault(title, []).append(seed)
|
raw.setdefault(title, []).append(seed)
|
||||||
_log(topic, f"Seed „{seed}“ als Subbaustein eingefügt ({title}) — Facts-Gate prüft")
|
_log(topic, f"Seed „{seed}“ als Subbaustein eingefügt ({title}) — Facts-Gate prüft")
|
||||||
|
|
||||||
|
# AFTER the seed guarantee: promoted/inserted seeds must not bypass the near-dup filter
|
||||||
|
await _dedup_subblocks(topic, raw) # near-dup filter per block (deterministic, no LLM)
|
||||||
|
|
||||||
if not raw:
|
if not raw:
|
||||||
# Finders ran but nothing survived the consensus/evidence gates: a legitimately
|
# Finders ran but nothing survived the consensus/evidence gates: a legitimately
|
||||||
# thin block (e.g. a bare named reduction). {} = done-without-subs — the guide
|
# thin block (e.g. a bare named reduction). {} = done-without-subs — the guide
|
||||||
@@ -1168,6 +1177,343 @@ async def _dedup_subblocks(topic: str, raw: dict[str, list[str]]) -> None:
|
|||||||
raw[title] = [subs[i] for i in sorted(keepers)] # original order of the kept ones
|
raw[title] = [subs[i] for i in sorted(keepers)] # original order of the kept ones
|
||||||
|
|
||||||
|
|
||||||
|
_KONSOLIDIERUNG_PANEL = 2 # merge needs unanimity — single judges over-merge (blocks-dedup lesson)
|
||||||
|
|
||||||
|
|
||||||
|
def _kons_id(x, n: int) -> int | None:
|
||||||
|
"""Judge id → int in 1..n, else None (bools are not ids)."""
|
||||||
|
if isinstance(x, bool):
|
||||||
|
return None
|
||||||
|
if isinstance(x, str) and x.isdigit():
|
||||||
|
x = int(x)
|
||||||
|
return x if isinstance(x, int) and 1 <= x <= n else None
|
||||||
|
|
||||||
|
|
||||||
|
def _konsolidierung_schema(data, n: int) -> dict | None:
|
||||||
|
"""Judge output → normalized dict, else None. gruppen accepts the {"haupt": 1,
|
||||||
|
"weitere": [4]} form AND the legacy plain-list form [1, 4] (resume files of the
|
||||||
|
first template version). kataloge/fremd/luecken are optional."""
|
||||||
|
if not isinstance(data, dict) or not isinstance(data.get("gruppen"), list):
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _ids(lst):
|
||||||
|
return sorted({i for x in (lst or []) if (i := _kons_id(x, n)) is not None})
|
||||||
|
|
||||||
|
gruppen = []
|
||||||
|
for g in data["gruppen"]:
|
||||||
|
if isinstance(g, dict):
|
||||||
|
haupt = _kons_id(g.get("haupt"), n)
|
||||||
|
ids = _ids(([haupt] if haupt else []) + list(g.get("weitere") or []))
|
||||||
|
elif isinstance(g, list):
|
||||||
|
haupt, ids = None, _ids(g)
|
||||||
|
else:
|
||||||
|
return None
|
||||||
|
if len(ids) >= 2:
|
||||||
|
gruppen.append({"haupt": haupt if haupt in ids else None, "ids": ids})
|
||||||
|
kataloge = []
|
||||||
|
for k in data.get("kataloge") or []:
|
||||||
|
if not isinstance(k, dict):
|
||||||
|
continue
|
||||||
|
ids = _ids(k.get("mitglieder"))
|
||||||
|
titel = str(k.get("titel") or "").strip()
|
||||||
|
if len(ids) >= 2 and titel:
|
||||||
|
kataloge.append({"titel": titel, "ids": ids})
|
||||||
|
return {"gruppen": gruppen, "kataloge": kataloge, "fremd": set(_ids(data.get("fremd"))),
|
||||||
|
"luecken": [s.strip() for s in data.get("luecken") or [] if isinstance(s, str) and s.strip()]}
|
||||||
|
|
||||||
|
|
||||||
|
def _facts_union(wf: dict, lf: dict) -> None:
|
||||||
|
"""Merge a folded sub's facts into the winner's: key_points/cited_facts union
|
||||||
|
(exact-duplicate-free), scalar fields only fill gaps."""
|
||||||
|
for feld in ("key_points", "cited_facts"):
|
||||||
|
have = wf.get(feld) or []
|
||||||
|
seen = {json.dumps(e, sort_keys=True, ensure_ascii=False) for e in have}
|
||||||
|
fresh = [e for e in (lf.get(feld) or [])
|
||||||
|
if json.dumps(e, sort_keys=True, ensure_ascii=False) not in seen]
|
||||||
|
if fresh:
|
||||||
|
wf[feld] = have + fresh
|
||||||
|
for feld in ("prerequisites", "hurdles", "example_idea"):
|
||||||
|
if not wf.get(feld) and lf.get(feld):
|
||||||
|
wf[feld] = lf[feld]
|
||||||
|
|
||||||
|
|
||||||
|
def _agreed_cliques(pair_sets: list[set], negs: list, n: int) -> list[list[int]]:
|
||||||
|
"""Union-find over the UNANIMOUS pairs (both judges grouped them), negation-guarded."""
|
||||||
|
agreed = {(a, b) for a, b in pair_sets[0] & pair_sets[1] if negs[a - 1] == negs[b - 1]}
|
||||||
|
parent = list(range(n + 1))
|
||||||
|
|
||||||
|
def find(x):
|
||||||
|
while parent[x] != x:
|
||||||
|
parent[x] = parent[parent[x]]
|
||||||
|
x = parent[x]
|
||||||
|
return x
|
||||||
|
|
||||||
|
for a, b in agreed:
|
||||||
|
parent[find(a)] = find(b)
|
||||||
|
groups: dict[int, list[int]] = {}
|
||||||
|
for k in range(1, n + 1):
|
||||||
|
groups.setdefault(find(k), []).append(k)
|
||||||
|
return [g for g in groups.values() if len(g) >= 2]
|
||||||
|
|
||||||
|
|
||||||
|
def _pairs_of(groups) -> set:
|
||||||
|
ps: set[tuple[int, int]] = set()
|
||||||
|
for ids in groups:
|
||||||
|
ps |= {(a, b) for x, a in enumerate(ids) for b in ids[x + 1:]}
|
||||||
|
return ps
|
||||||
|
|
||||||
|
|
||||||
|
_LUECKEN_CAP = 3 # the gap list feeds ONE finder round — an uncapped list doubled the decomposition
|
||||||
|
|
||||||
|
|
||||||
|
def _luecken_schnitt(l1: list[str], l2: list[str], cap: int = _LUECKEN_CAP) -> list[str]:
|
||||||
|
"""Gaps BOTH judges name — exact strings never match across paraphrases, so a gap
|
||||||
|
survives when the other judge names one sharing a distinctive token (≥4 chars).
|
||||||
|
j1's phrasing wins. The measured union produced 107 'gaps' on 216 subs."""
|
||||||
|
def toks(s):
|
||||||
|
return {t for t in _sub_tokens(s) if len(t) >= 4}
|
||||||
|
toks2 = [toks(l) for l in l2]
|
||||||
|
out = [l for l in l1 if toks(l) and any(toks(l) & t2 for t2 in toks2)]
|
||||||
|
return out[:cap]
|
||||||
|
|
||||||
|
|
||||||
|
async def _konsolidiere_subblocks(ctx: GenContext, files: dict, raw: dict, facts_map: dict,
|
||||||
|
instructions: str = "", ns: str = "", lbl: str = "") -> dict:
|
||||||
|
"""In-block consolidation AFTER the facts stage: a two-judge panel sees the subs WITH
|
||||||
|
their key points and applies the 100%-decomposition test — the embedding paths only
|
||||||
|
catch cos ≥ 0.90, real paraphrase duplicates measure down to 0.61, and only the facts
|
||||||
|
reveal a subset. Every action needs UNANIMITY of both judges:
|
||||||
|
gruppen — same-statement/subset entries fold into the judge-named `haupt` (base
|
||||||
|
before detail; heuristic fallback), facts union, losers → `variant`
|
||||||
|
kataloge — pure enumeration entries of one kind bundle into a NEW named sub row
|
||||||
|
(members → `variant`); runs before levels/relevance, so the new row
|
||||||
|
gets classified normally
|
||||||
|
fremd — statements off-topic for the TOPIC → `discarded` (removal test)
|
||||||
|
Questions/artefacts do not exist yet — no orphans. Gaps are returned per block so the
|
||||||
|
caller can run the single follow-up finder round (`_luecken_runde`).
|
||||||
|
Judge replies persist as j-files keyed by a subs-list hash (resume-safe).
|
||||||
|
→ {block title: [luecken]}"""
|
||||||
|
topic = ctx.topic
|
||||||
|
work_dir = files["arbeit"]
|
||||||
|
luecken_by_title: dict[str, list[str]] = {}
|
||||||
|
for title, subs in list(raw.items()):
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return luecken_by_title
|
||||||
|
n = len(subs)
|
||||||
|
if n < 2:
|
||||||
|
continue
|
||||||
|
bnorm = _norm_title(title)
|
||||||
|
bfacts = facts_map.setdefault(title, {})
|
||||||
|
|
||||||
|
def _kp(s):
|
||||||
|
return (bfacts.get(_norm_title(s)) or {}).get("key_points") or []
|
||||||
|
|
||||||
|
# prompt shows max 3 key points per sub — full lists blew past the judge timeout
|
||||||
|
# (measured: 15 % timeouts at 585 s); the facts UNION on merge stays complete
|
||||||
|
lines = "\n".join(f"{k}. {s}" + "".join(f"\n - {p}" for p in _kp(s)[:3])
|
||||||
|
for k, s in enumerate(subs, 1))
|
||||||
|
h = hashlib.md5("\n".join(subs).encode()).hexdigest()[:8]
|
||||||
|
paths = [work_dir / f"sub-konsolidierung-{ns}{h}-j{j}.json" for j in (1, 2)]
|
||||||
|
|
||||||
|
async def _judge(j, path):
|
||||||
|
if _konsolidierung_schema(_json_file(path), n) is not None:
|
||||||
|
return # resume
|
||||||
|
status, _v = await run_single_slot(
|
||||||
|
ctx, f"{lbl}Sub-Konsolidierung j{j}",
|
||||||
|
key=f"blocks-{topic}-{ns}sub-konsolidierung-{h}-j{j}",
|
||||||
|
prompt=_prompt("Subblock-Konsolidierung", topic=topic, block=title, subs=lines, extra=_extra(instructions)),
|
||||||
|
role="judge", capabilities="none",
|
||||||
|
payload=lambda result, p=path: _sink_json(result, p, lambda d: _konsolidierung_schema(d, n)),
|
||||||
|
timeout=_timeout("konsolidierung", n))
|
||||||
|
if status == FAILED:
|
||||||
|
_log(topic, f"Sub-Konsolidierung {title} j{j} ohne Ergebnis — fail-open")
|
||||||
|
|
||||||
|
await asyncio.gather(*[_judge(j, p) for j, p in zip((1, 2), paths)])
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return luecken_by_title
|
||||||
|
outs = [o for p in paths if (o := _konsolidierung_schema(_json_file(p), n)) is not None]
|
||||||
|
if len(outs) == 1: # Ersatz-Richter: EIN Timeout darf die gute Stimme nicht entwerten
|
||||||
|
ersatz = work_dir / f"sub-konsolidierung-{ns}{h}-j3.json"
|
||||||
|
await _judge(3, ersatz)
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return luecken_by_title
|
||||||
|
outs = [o for p in [*paths, ersatz]
|
||||||
|
if (o := _konsolidierung_schema(_json_file(p), n)) is not None]
|
||||||
|
# gaps need UNANIMITY (token-overlap match) — the union of both judges was uncalibrated
|
||||||
|
luecken = (_luecken_schnitt(outs[0]["luecken"], outs[1]["luecken"])
|
||||||
|
if len(outs) == _KONSOLIDIERUNG_PANEL else [])
|
||||||
|
journal = {"block": title, "richter": len(outs), "vorher": n,
|
||||||
|
"luecken_roh": [len(o["luecken"]) for o in outs],
|
||||||
|
"gruppen": [], "kataloge": [], "fremd": [], "luecken": luecken}
|
||||||
|
if len(outs) == _KONSOLIDIERUNG_PANEL:
|
||||||
|
negs = [_neg_set(s) for s in subs]
|
||||||
|
keep = list(subs)
|
||||||
|
gone: set[int] = set()
|
||||||
|
|
||||||
|
async def _fold(k: int, wf: dict | None):
|
||||||
|
lose_title = subs[k - 1]
|
||||||
|
lf = bfacts.pop(_norm_title(lose_title), None) or {}
|
||||||
|
if wf is not None:
|
||||||
|
_facts_union(wf, lf)
|
||||||
|
await db.set_subblock_fields(topic, bnorm, _norm_title(lose_title), status="variant")
|
||||||
|
keep.remove(lose_title)
|
||||||
|
gone.add(k)
|
||||||
|
|
||||||
|
# 1. Fremd (removal test): off-topic for the TOPIC → discarded, no heir.
|
||||||
|
for k in sorted(outs[0]["fremd"] & outs[1]["fremd"]):
|
||||||
|
ft = subs[k - 1]
|
||||||
|
bfacts.pop(_norm_title(ft), None)
|
||||||
|
await db.set_subblock_fields(topic, bnorm, _norm_title(ft), status="discarded")
|
||||||
|
keep.remove(ft)
|
||||||
|
gone.add(k)
|
||||||
|
journal["fremd"].append(ft)
|
||||||
|
|
||||||
|
# 2. Gruppen: winner = judge-named haupt (majority), else key_points/length heuristic.
|
||||||
|
haupt_votes: dict[int, int] = {}
|
||||||
|
for o in outs:
|
||||||
|
for g in o["gruppen"]:
|
||||||
|
if g["haupt"]:
|
||||||
|
haupt_votes[g["haupt"]] = haupt_votes.get(g["haupt"], 0) + 1
|
||||||
|
for g in _agreed_cliques([_pairs_of([x["ids"] for x in o["gruppen"]]) for o in outs], negs, n):
|
||||||
|
g = [k for k in g if k not in gone]
|
||||||
|
if len(g) < 2:
|
||||||
|
continue
|
||||||
|
win = max(g, key=lambda k: (haupt_votes.get(k, 0),
|
||||||
|
len(_kp(subs[k - 1])), len(subs[k - 1]), -k))
|
||||||
|
wf = bfacts.setdefault(_norm_title(subs[win - 1]), {})
|
||||||
|
for k in g:
|
||||||
|
if k != win:
|
||||||
|
await _fold(k, wf)
|
||||||
|
journal["gruppen"].append({"behalten": subs[win - 1],
|
||||||
|
"gefaltet": [subs[k - 1] for k in g if k != win]})
|
||||||
|
|
||||||
|
# 3. Kataloge: bundle enumeration rows into ONE new named sub (facts union).
|
||||||
|
for g in _agreed_cliques([_pairs_of([x["ids"] for x in o["kataloge"]]) for o in outs], negs, n):
|
||||||
|
g = [k for k in g if k not in gone]
|
||||||
|
if len(g) < 2:
|
||||||
|
continue
|
||||||
|
titel = next((clean_title(x["titel"]) for x in outs[0]["kataloge"] + outs[1]["kataloge"]
|
||||||
|
if set(x["ids"]) & set(g) and clean_title(x["titel"])), "")
|
||||||
|
kn = _norm_title(titel)
|
||||||
|
if not kn or kn in {_norm_title(s) for s in keep}:
|
||||||
|
continue # no usable/colliding title → members stay
|
||||||
|
kf: dict = {}
|
||||||
|
for k in g:
|
||||||
|
await _fold(k, kf)
|
||||||
|
bfacts[kn] = kf
|
||||||
|
keep.append(titel)
|
||||||
|
await db.put_subblock(topic, bnorm, kn, title, titel, status="consensus")
|
||||||
|
journal["kataloge"].append({"titel": titel, "gefaltet": [subs[k - 1] for k in g]})
|
||||||
|
|
||||||
|
if len(keep) != n:
|
||||||
|
raw[title] = keep
|
||||||
|
_log(topic, f"Sub-Konsolidierung {title}: {n} → {len(keep)}")
|
||||||
|
elif outs:
|
||||||
|
_log(topic, f"Sub-Konsolidierung {title}: nur {len(outs)}/{_KONSOLIDIERUNG_PANEL} Richter — fail-open")
|
||||||
|
if luecken:
|
||||||
|
luecken_by_title[title] = luecken
|
||||||
|
_log(topic, f"Sub-Konsolidierung {title}: mögliche Lücken: {', '.join(luecken[:5])}")
|
||||||
|
atomic_write_json(work_dir / f"sub-konsolidierung-{ns}{h}.json", journal, indent=1)
|
||||||
|
return luecken_by_title
|
||||||
|
|
||||||
|
|
||||||
|
async def _luecken_runde(ctx: GenContext, files: dict, title: str, luecken: list[str],
|
||||||
|
raw: dict, facts_map: dict, q: dict, folder, instructions: str = "",
|
||||||
|
ns: str = "", lbl: str = "", sources: list[str] | None = None) -> int:
|
||||||
|
"""ONE targeted finder round for the consolidation judges' reported gaps — no loop.
|
||||||
|
Finds are deduped against the existing subs (token containment + embedding +
|
||||||
|
negation guard, seed-guarantee pattern) and must pass the facts evidence gate
|
||||||
|
(own work subdir `nf` — the block's facts resume files must not collide) before
|
||||||
|
they join raw/facts_map as consensus rows. They then flow through levels/relevance/
|
||||||
|
questions/artefacts like any other sub. → count of adopted subs."""
|
||||||
|
topic, provider, is_cancelled = ctx.topic, ctx.provider, ctx.is_cancelled
|
||||||
|
work_dir = files["arbeit"]
|
||||||
|
have = list(raw.get(title) or [])
|
||||||
|
focus = (instructions + "\n\nFinde NUR belegbare Subbausteine zu diesen bisher fehlenden "
|
||||||
|
"Aspekten des Blocks — nichts anderes:\n" + "\n".join(f"- {l}" for l in luecken))
|
||||||
|
known = ("\n\nBEREITS ERFASST — liste diese NICHT erneut:\n"
|
||||||
|
+ "\n".join(f"- {s}" for s in have)) if have else ""
|
||||||
|
paths = [work_dir / f"luecken-{ns}r1-{i}.md" for i in (1, 2, 3)]
|
||||||
|
slots = [{
|
||||||
|
"key": f"blocks-{topic}-{ns}luecken-r1-{i}",
|
||||||
|
"prompt": _prompt("Subblock-Research", topic=topic, assignment=f"- {title}", known=known, out_path=p, extra=_extra(focus)),
|
||||||
|
"role": "quick", "capabilities": "files" if folder else "full",
|
||||||
|
"payload": (lambda result, p=p: _parse_subblocks(_read(p)) or None),
|
||||||
|
} for i, p in zip((1, 2, 3), paths)]
|
||||||
|
agent_texts = await _race(topic, f"{lbl}Lücken-Nachfass", slots, 2,
|
||||||
|
_timeout("subblock", 1), provider, cancelled=is_cancelled, grace=CONSENSUS_GRACE)
|
||||||
|
if is_cancelled() or not agent_texts:
|
||||||
|
return 0
|
||||||
|
cands: list[str] = []
|
||||||
|
seen = {_norm_title(s) for s in have}
|
||||||
|
for d in agent_texts:
|
||||||
|
for subs in d.values(): # single-block call — every marker means this block
|
||||||
|
for s in subs:
|
||||||
|
sn = _norm_title(s)
|
||||||
|
if sn and sn not in seen:
|
||||||
|
seen.add(sn)
|
||||||
|
cands.append(s)
|
||||||
|
if not cands:
|
||||||
|
return 0
|
||||||
|
emb_on = EMBEDDING_AKTIV and await asyncio.to_thread(embedding.available)
|
||||||
|
fresh: list[str] = []
|
||||||
|
for s in cands:
|
||||||
|
st = _sub_tokens(s)
|
||||||
|
base = have + fresh
|
||||||
|
if any(st <= _sub_tokens(b) or _sub_tokens(b) <= st for b in base):
|
||||||
|
continue
|
||||||
|
if emb_on and base:
|
||||||
|
sims = await asyncio.to_thread(embedding.embed_sims, [s] + base)
|
||||||
|
if sims is not None:
|
||||||
|
negs = [_neg_set(t) for t in [s] + base]
|
||||||
|
if any(float(sims[0][j]) >= SEED_COVER_COS and negs[0] == negs[j]
|
||||||
|
for j in range(1, len(base) + 1)):
|
||||||
|
continue
|
||||||
|
fresh.append(s)
|
||||||
|
if not fresh:
|
||||||
|
return 0
|
||||||
|
nf_dir = work_dir / "nf"
|
||||||
|
nf_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
res = await _facts_block(ctx, lambda *a, **k: None, {**files, "arbeit": nf_dir},
|
||||||
|
{title: list(fresh)}, q, folder, instructions,
|
||||||
|
ns=f"{ns}nf-", lbl=lbl, sources=sources, slim=True)
|
||||||
|
if is_cancelled() or res is None:
|
||||||
|
return 0
|
||||||
|
nf_facts, discarded = res
|
||||||
|
dropped = (discarded or {}).get(title) or set()
|
||||||
|
nf_map = nf_facts.get(title) or {}
|
||||||
|
|
||||||
|
def _belegt(s: str) -> bool: # HARD gate: no facts entry = no evidence = no adoption
|
||||||
|
fk = nf_map.get(_norm_title(s))
|
||||||
|
return bool(fk and (fk.get("key_points") or fk.get("cited_facts")))
|
||||||
|
|
||||||
|
kept = [s for s in fresh if _norm_title(s) not in dropped and _belegt(s)]
|
||||||
|
if not kept:
|
||||||
|
return 0
|
||||||
|
bnorm = _norm_title(title)
|
||||||
|
bfacts = facts_map.setdefault(title, {})
|
||||||
|
for s in kept:
|
||||||
|
sn = _norm_title(s)
|
||||||
|
await db.upsert_subblock(topic, bnorm, sn, title, s)
|
||||||
|
await db.set_subblock_fields(topic, bnorm, sn, status="consensus")
|
||||||
|
bfacts[sn] = nf_map[sn]
|
||||||
|
raw.setdefault(title, []).extend(kept)
|
||||||
|
_log(topic, f"Lücken-Nachfass {title}: {len(kept)}/{len(fresh)} Funde übernommen")
|
||||||
|
return len(kept)
|
||||||
|
|
||||||
|
|
||||||
|
def _subs_hash(sidecar_or_raw: dict) -> str:
|
||||||
|
"""Sub-set identity for the resume files of the sub-CONSUMING stages (levels/relevance/
|
||||||
|
questions/artefacts). Without it a re-run with a recut sub set adopted the stale stage
|
||||||
|
results (measured: 626 orphans — artefacts of the old 425-sub set re-imported)."""
|
||||||
|
parts: list[str] = []
|
||||||
|
for title, subs in sidecar_or_raw.items():
|
||||||
|
parts.append(str(title))
|
||||||
|
for s in subs:
|
||||||
|
parts.append(s["title"] if isinstance(s, dict) else str(s))
|
||||||
|
return hashlib.md5("\n".join(parts).encode()).hexdigest()[:8]
|
||||||
|
|
||||||
|
|
||||||
def _code_vote(rater: list[dict], n: int) -> tuple[dict, dict]:
|
def _code_vote(rater: list[dict], n: int) -> tuple[dict, dict]:
|
||||||
"""Majority vote over rater dicts on local ids 1..n → (outcome, disputed). A clear winner
|
"""Majority vote over rater dicts on local ids 1..n → (outcome, disputed). A clear winner
|
||||||
needs ≥2 votes and no tie; otherwise the id is disputed (kept with its vote list)."""
|
needs ≥2 votes and no tie; otherwise the id is disputed (kept with its vote list)."""
|
||||||
@@ -1220,9 +1566,10 @@ async def _levels_block(ctx: GenContext, set_p, files: dict, raw: dict, instruct
|
|||||||
if cur:
|
if cur:
|
||||||
chunks.append(cur)
|
chunks.append(cur)
|
||||||
n = len(chunks)
|
n = len(chunks)
|
||||||
|
sh = _subs_hash(raw) # resume must invalidate when the sub set changed
|
||||||
|
|
||||||
def rater_paths(c):
|
def rater_paths(c):
|
||||||
return [work_dir / f"level-c{c}-{i}.json" for i in (1, 2, 3)]
|
return [work_dir / f"level-{sh}-c{c}-{i}.json" for i in (1, 2, 3)]
|
||||||
|
|
||||||
def lset(item_idxs):
|
def lset(item_idxs):
|
||||||
return set(range(1, len(item_idxs) + 1))
|
return set(range(1, len(item_idxs) + 1))
|
||||||
@@ -1273,7 +1620,7 @@ async def _levels_block(ctx: GenContext, set_p, files: dict, raw: dict, instruct
|
|||||||
async def _clarify(c, item_idxs):
|
async def _clarify(c, item_idxs):
|
||||||
outcome, strittig = vote_by_c[c]
|
outcome, strittig = vote_by_c[c]
|
||||||
if strittig:
|
if strittig:
|
||||||
judge_path = work_dir / f"level-final-c{c}.json"
|
judge_path = work_dir / f"level-final-{sh}-c{c}.json"
|
||||||
decision = _levels_schema(_json_file(judge_path), set(strittig))
|
decision = _levels_schema(_json_file(judge_path), set(strittig))
|
||||||
if decision is None:
|
if decision is None:
|
||||||
disputed_block = _disputed_lines(items, item_idxs, strittig)
|
disputed_block = _disputed_lines(items, item_idxs, strittig)
|
||||||
@@ -1385,9 +1732,11 @@ def _facts_lines(fk: dict) -> str:
|
|||||||
return "\n".join(z)
|
return "\n".join(z)
|
||||||
|
|
||||||
|
|
||||||
async def _facts_block(ctx, set_p, files: dict, raw: dict, q: dict, folder, instructions: str, ns: str = "", lbl: str = "", sources: list[str] | None = None) -> tuple | None:
|
async def _facts_block(ctx, set_p, files: dict, raw: dict, q: dict, folder, instructions: str, ns: str = "", lbl: str = "", sources: list[str] | None = None, slim: bool = False) -> tuple | None:
|
||||||
"""Block: per sub extract source facts (find) → verify (check) → correct/discard (fix).
|
"""Block: per sub extract source facts (find) → verify (check) → correct/discard (fix).
|
||||||
Extract-once grounding: the result feeds level/relevance/questions/guide.
|
Extract-once grounding: the result feeds level/relevance/questions/guide.
|
||||||
|
slim=True (gap follow-up): no supplement pass, ONE check judge — the full program cost
|
||||||
|
230 agent-minutes per run for a handful of finds; the hard adoption gate stays.
|
||||||
→ (facts_map, discarded_map) — facts_map {block: {sub_norm: facts}}, discarded_map
|
→ (facts_map, discarded_map) — facts_map {block: {sub_norm: facts}}, discarded_map
|
||||||
{block: {sub_norm}} (unsupportable subs to remove) — or None on cancel/error."""
|
{block: {sub_norm}} (unsupportable subs to remove) — or None on cancel/error."""
|
||||||
topic, provider, is_cancelled = ctx.topic, ctx.provider, ctx.is_cancelled
|
topic, provider, is_cancelled = ctx.topic, ctx.provider, ctx.is_cancelled
|
||||||
@@ -1399,11 +1748,12 @@ async def _facts_block(ctx, set_p, files: dict, raw: dict, q: dict, folder, inst
|
|||||||
if not blocks:
|
if not blocks:
|
||||||
return {}, {}
|
return {}, {}
|
||||||
chunks = _lpt_chunks([len(subs) for _, subs in blocks], FACTS_CHUNK_SUBS)
|
chunks = _lpt_chunks([len(subs) for _, subs in blocks], FACTS_CHUNK_SUBS)
|
||||||
|
sh = _subs_hash(raw) # resume must invalidate when the sub set changed
|
||||||
|
|
||||||
def raw_path(ci): return work_dir / f"facts-c{ci}.json"
|
def raw_path(ci): return work_dir / f"facts-{sh}-c{ci}.json"
|
||||||
def supp_path(ci): return work_dir / f"facts-erg-c{ci}.json"
|
def supp_path(ci): return work_dir / f"facts-erg-{sh}-c{ci}.json"
|
||||||
def chk_path(ci, j): return work_dir / f"facts-check-c{ci}-j{j}.json"
|
def chk_path(ci, j): return work_dir / f"facts-check-{sh}-c{ci}-j{j}.json"
|
||||||
def fix_path(ci): return work_dir / f"facts-fix-c{ci}.json"
|
def fix_path(ci): return work_dir / f"facts-fix-{sh}-c{ci}.json"
|
||||||
def ctitle(idxs): return [blocks[i][0] for i in idxs]
|
def ctitle(idxs): return [blocks[i][0] for i in idxs]
|
||||||
def block_text(idxs):
|
def block_text(idxs):
|
||||||
return "\n\n".join(
|
return "\n\n".join(
|
||||||
@@ -1495,10 +1845,13 @@ async def _facts_block(ctx, set_p, files: dict, raw: dict, q: dict, folder, inst
|
|||||||
payload=lambda result, p=ep: _facts_schema(_json_file(p)),
|
payload=lambda result, p=ep: _facts_schema(_json_file(p)),
|
||||||
timeout=_timeout("content", subs_total))
|
timeout=_timeout("content", subs_total))
|
||||||
|
|
||||||
set_p("Facts supplement…", step=_step_idx(topic, "Facts find"))
|
if not slim:
|
||||||
await _gather_progress([_supplement(ci, idxs) for ci, idxs in enumerate(chunks)], len(chunks), _report_p(set_p, topic, "Facts find"))
|
set_p("Facts supplement…", step=_step_idx(topic, "Facts find"))
|
||||||
if is_cancelled():
|
await _gather_progress([_supplement(ci, idxs) for ci, idxs in enumerate(chunks)], len(chunks), _report_p(set_p, topic, "Facts find"))
|
||||||
return None
|
if is_cancelled():
|
||||||
|
return None
|
||||||
|
panel = (1,) if slim else (1, 2, 3)[:FACTS_CHECK_PANEL]
|
||||||
|
min_discard = 1 if slim else 2
|
||||||
|
|
||||||
# Phase "Facts check": FACTS_CHECK_PANEL judges per chunk. Two majority sets:
|
# Phase "Facts check": FACTS_CHECK_PANEL judges per chunk. Two majority sets:
|
||||||
# flagged (fact inaccurate → correct) and discard (sub not supportable → remove).
|
# flagged (fact inaccurate → correct) and discard (sub not supportable → remove).
|
||||||
@@ -1514,7 +1867,7 @@ async def _facts_block(ctx, set_p, files: dict, raw: dict, q: dict, folder, inst
|
|||||||
fallback = [blocks[i][0] for i in idxs] + [fk["sub"] for fm in per.values() for fk in fm.values()]
|
fallback = [blocks[i][0] for i in idxs] + [fk["sub"] for fm in per.values() for fk in fm.values()]
|
||||||
ev = _cited_evidence(folder, sources, cites, fallback) if folder else ""
|
ev = _cited_evidence(folder, sources, cites, fallback) if folder else ""
|
||||||
c_source = _prompt("Blocks-Source-Inline", excerpts=ev) if ev else source
|
c_source = _prompt("Blocks-Source-Inline", excerpts=ev) if ev else source
|
||||||
pending = [j for j in (1, 2, 3)[:FACTS_CHECK_PANEL] if _facts_check_schema(_json_file(chk_path(ci, j))) is None]
|
pending = [j for j in panel if _facts_check_schema(_json_file(chk_path(ci, j))) is None]
|
||||||
rs = await asyncio.gather(*[
|
rs = await asyncio.gather(*[
|
||||||
run_agent(f"blocks-{topic}-{ns}facts-check-c{ci}-j{j}",
|
run_agent(f"blocks-{topic}-{ns}facts-check-c{ci}-j{j}",
|
||||||
_prompt("Facts-Check", topic=topic, source=c_source, facts=facts_text, out_path=chk_path(ci, j), extra=_extra(instructions)),
|
_prompt("Facts-Check", topic=topic, source=c_source, facts=facts_text, out_path=chk_path(ci, j), extra=_extra(instructions)),
|
||||||
@@ -1525,7 +1878,7 @@ async def _facts_block(ctx, set_p, files: dict, raw: dict, q: dict, folder, inst
|
|||||||
for j, r in zip(pending, rs):
|
for j, r in zip(pending, rs):
|
||||||
if isinstance(r, tuple):
|
if isinstance(r, tuple):
|
||||||
_sink_json(r, chk_path(ci, j), _facts_check_schema)
|
_sink_json(r, chk_path(ci, j), _facts_check_schema)
|
||||||
outs = [s for j in (1, 2, 3)[:FACTS_CHECK_PANEL] if (s := _facts_check_schema(_json_file(chk_path(ci, j)))) is not None]
|
outs = [s for j in panel if (s := _facts_check_schema(_json_file(chk_path(ci, j)))) is not None]
|
||||||
bvotes: dict[str, int] = {}
|
bvotes: dict[str, int] = {}
|
||||||
vvotes: dict[str, int] = {}
|
vvotes: dict[str, int] = {}
|
||||||
for s in outs: # s = [(sub_norm, verwerfen)] of one judge
|
for s in outs: # s = [(sub_norm, verwerfen)] of one judge
|
||||||
@@ -1538,8 +1891,9 @@ async def _facts_block(ctx, set_p, files: dict, raw: dict, q: dict, folder, inst
|
|||||||
threshold = len(outs) / 2 if outs else 99
|
threshold = len(outs) / 2 if outs else 99
|
||||||
flagged = {sn for sn, v in bvotes.items() if v > threshold}
|
flagged = {sn for sn, v in bvotes.items() if v > threshold}
|
||||||
# Discarding is irreversible → stricter than flagging: majority AND ≥2 agreeing judges
|
# Discarding is irreversible → stricter than flagging: majority AND ≥2 agreeing judges
|
||||||
# (prevents deletion by a single vote when the panel is degraded).
|
# (prevents deletion by a single vote when the panel is degraded). slim runs ONE judge
|
||||||
to_discard = {sn for sn, v in vvotes.items() if v > threshold and v >= 2}
|
# by design — there its single vote must be allowed to discard.
|
||||||
|
to_discard = {sn for sn, v in vvotes.items() if v > threshold and v >= min_discard}
|
||||||
return ci, flagged, to_discard
|
return ci, flagged, to_discard
|
||||||
|
|
||||||
check = await _gather_progress([_check(ci, idxs) for ci, idxs in enumerate(chunks)], len(chunks), _report_p(set_p, topic, "Facts check"))
|
check = await _gather_progress([_check(ci, idxs) for ci, idxs in enumerate(chunks)], len(chunks), _report_p(set_p, topic, "Facts check"))
|
||||||
@@ -1611,9 +1965,10 @@ async def _relevance_block(ctx: GenContext, set_p, files: dict, sidecar: dict, i
|
|||||||
return {}
|
return {}
|
||||||
chunks = _chunk_nums(list(range(len(items))), _n_chunks(len(items), LEVEL_CHUNK))
|
chunks = _chunk_nums(list(range(len(items))), _n_chunks(len(items), LEVEL_CHUNK))
|
||||||
n = len(chunks)
|
n = len(chunks)
|
||||||
|
sh = _subs_hash(sidecar) # resume must invalidate when the sub set changed
|
||||||
|
|
||||||
def rater_paths(c):
|
def rater_paths(c):
|
||||||
return [work_dir / f"relevance-c{c}-{i}.json" for i in (1, 2, 3)]
|
return [work_dir / f"relevance-{sh}-c{c}-{i}.json" for i in (1, 2, 3)]
|
||||||
|
|
||||||
def lset(item_idxs):
|
def lset(item_idxs):
|
||||||
return set(range(1, len(item_idxs) + 1))
|
return set(range(1, len(item_idxs) + 1))
|
||||||
@@ -1660,7 +2015,7 @@ async def _relevance_block(ctx: GenContext, set_p, files: dict, sidecar: dict, i
|
|||||||
async def _clarify(c, item_idxs):
|
async def _clarify(c, item_idxs):
|
||||||
outcome, strittig = vote_by_c[c]
|
outcome, strittig = vote_by_c[c]
|
||||||
if strittig:
|
if strittig:
|
||||||
judge_path = work_dir / f"relevance-final-c{c}.json"
|
judge_path = work_dir / f"relevance-final-{sh}-c{c}.json"
|
||||||
decision = _relevance_schema(_json_file(judge_path), set(strittig))
|
decision = _relevance_schema(_json_file(judge_path), set(strittig))
|
||||||
if decision is None:
|
if decision is None:
|
||||||
disputed_block = _disputed_lines(items, item_idxs, strittig)
|
disputed_block = _disputed_lines(items, item_idxs, strittig)
|
||||||
@@ -1733,12 +2088,13 @@ async def _question_pattern_block(ctx: GenContext, set_p, files: dict, sidecar:
|
|||||||
if not blocks:
|
if not blocks:
|
||||||
return {}
|
return {}
|
||||||
chunks = _lpt_chunks([len(rel) for _, rel in blocks], QUESTION_CHUNK_SUBS) # load-balanced by sub count
|
chunks = _lpt_chunks([len(rel) for _, rel in blocks], QUESTION_CHUNK_SUBS) # load-balanced by sub count
|
||||||
|
sh = _subs_hash(sidecar) # resume must invalidate when the sub set changed
|
||||||
|
|
||||||
def raw_path(ci):
|
def raw_path(ci):
|
||||||
return work_dir / f"question-pattern-c{ci}.json"
|
return work_dir / f"question-pattern-{sh}-c{ci}.json"
|
||||||
|
|
||||||
def final_path(ci):
|
def final_path(ci):
|
||||||
return work_dir / f"question-pattern-final-c{ci}.json"
|
return work_dir / f"question-pattern-final-{sh}-c{ci}.json"
|
||||||
|
|
||||||
def _chunk_title(idxs):
|
def _chunk_title(idxs):
|
||||||
return [blocks[i][0] for i in idxs]
|
return [blocks[i][0] for i in idxs]
|
||||||
@@ -1889,7 +2245,7 @@ async def _question_pattern_block(ctx: GenContext, set_p, files: dict, sidecar:
|
|||||||
return "\n\n".join(block_texts)
|
return "\n\n".join(block_texts)
|
||||||
|
|
||||||
async def _request_more(round_n, pi, items):
|
async def _request_more(round_n, pi, items):
|
||||||
fp = work_dir / f"question-pattern-nach{round_n}-c{pi}.json"
|
fp = work_dir / f"question-pattern-nach{round_n}-{sh}-c{pi}.json"
|
||||||
if _question_pattern_chunk_schema(_json_file(fp)):
|
if _question_pattern_chunk_schema(_json_file(fp)):
|
||||||
return # resume
|
return # resume
|
||||||
subs_total = sum(len(s) for _, s in items)
|
subs_total = sum(len(s) for _, s in items)
|
||||||
@@ -1920,7 +2276,7 @@ async def _question_pattern_block(ctx: GenContext, set_p, files: dict, sidecar:
|
|||||||
for pi, items in enumerate(package_items):
|
for pi, items in enumerate(package_items):
|
||||||
title_subs = {t: subs for t, subs in items}
|
title_subs = {t: subs for t, subs in items}
|
||||||
ctitle = list(title_subs.keys())
|
ctitle = list(title_subs.keys())
|
||||||
for e in _question_pattern_chunk_schema(_json_file(work_dir / f"question-pattern-nach{round_n}-c{pi}.json")) or []:
|
for e in _question_pattern_chunk_schema(_json_file(work_dir / f"question-pattern-nach{round_n}-{sh}-c{pi}.json")) or []:
|
||||||
title = _match_sub(e["block"], ctitle)
|
title = _match_sub(e["block"], ctitle)
|
||||||
if title not in title_subs:
|
if title not in title_subs:
|
||||||
continue
|
continue
|
||||||
@@ -2825,6 +3181,7 @@ async def _artefacts_block(ctx: GenContext, set_p, files: dict, sidecar: dict, i
|
|||||||
topic, provider, is_cancelled = ctx.topic, ctx.provider, ctx.is_cancelled
|
topic, provider, is_cancelled = ctx.topic, ctx.provider, ctx.is_cancelled
|
||||||
work_dir = files["arbeit"]
|
work_dir = files["arbeit"]
|
||||||
caps = "files"
|
caps = "files"
|
||||||
|
sh = _subs_hash(sidecar) # resume must invalidate when the sub set changed
|
||||||
# Blocks with subs + facts lines as input block (extract-once from the facts).
|
# Blocks with subs + facts lines as input block (extract-once from the facts).
|
||||||
blocks = []
|
blocks = []
|
||||||
for btitle, subs in sidecar.items():
|
for btitle, subs in sidecar.items():
|
||||||
@@ -2855,7 +3212,7 @@ async def _artefacts_block(ctx: GenContext, set_p, files: dict, sidecar: dict, i
|
|||||||
async def _check_examples(ci, idxs, items):
|
async def _check_examples(ci, idxs, items):
|
||||||
if is_cancelled() or not items:
|
if is_cancelled() or not items:
|
||||||
return items
|
return items
|
||||||
def cpath(j): return work_dir / f"artifact-example-check-c{ci}-j{j}.json"
|
def cpath(j): return work_dir / f"artifact-example-check-{sh}-c{ci}-j{j}.json"
|
||||||
examples_txt = "\n\n".join(
|
examples_txt = "\n\n".join(
|
||||||
f"{k}. PROBLEM: {e['problem']}\n SCHRITTE: " + " | ".join(e.get("steps", []))
|
f"{k}. PROBLEM: {e['problem']}\n SCHRITTE: " + " | ".join(e.get("steps", []))
|
||||||
+ (f"\n ERGEBNIS: {e['result']}" if e.get("result") else "")
|
+ (f"\n ERGEBNIS: {e['result']}" if e.get("result") else "")
|
||||||
@@ -2890,7 +3247,7 @@ async def _artefacts_block(ctx: GenContext, set_p, files: dict, sidecar: dict, i
|
|||||||
own files) and run in parallel."""
|
own files) and run in parallel."""
|
||||||
schema = _ARTEFACT_SCHEMA[typ]
|
schema = _ARTEFACT_SCHEMA[typ]
|
||||||
|
|
||||||
def apath(ci): return work_dir / f"artifact-{typ}-c{ci}.json"
|
def apath(ci): return work_dir / f"artifact-{typ}-{sh}-c{ci}.json"
|
||||||
|
|
||||||
async def _gen(ci, idxs):
|
async def _gen(ci, idxs):
|
||||||
p = apath(ci)
|
p = apath(ci)
|
||||||
@@ -2945,7 +3302,7 @@ async def _mirror_sidecar_db(topic: str, sidecar: dict) -> None:
|
|||||||
|
|
||||||
|
|
||||||
async def generate_blocks(topic: str, instructions: str = "", provider: str = DEFAULT_PROVIDER,
|
async def generate_blocks(topic: str, instructions: str = "", provider: str = DEFAULT_PROVIDER,
|
||||||
research: bool = True) -> None:
|
research: bool = True, qa_force: bool = False) -> None:
|
||||||
"""Kanban entry point: source prep, then both boards (inventory + artefacts) until
|
"""Kanban entry point: source prep, then both boards (inventory + artefacts) until
|
||||||
quiescence. research=False = Continue (drain the existing queue, no new search).
|
quiescence. research=False = Continue (drain the existing queue, no new search).
|
||||||
A run on a finished topic ADDS research (live extension) — full rebuild = DELETE /blocks."""
|
A run on a finished topic ADDS research (live extension) — full rebuild = DELETE /blocks."""
|
||||||
@@ -2978,7 +3335,7 @@ async def generate_blocks(topic: str, instructions: str = "", provider: str = DE
|
|||||||
return
|
return
|
||||||
import board_inventory # lazy: the boards import blocks
|
import board_inventory # lazy: the boards import blocks
|
||||||
ok = await board_inventory.run_boards(ctx, set_p, files, q, folder, instructions,
|
ok = await board_inventory.run_boards(ctx, set_p, files, q, folder, instructions,
|
||||||
research=research)
|
research=research, qa_force=qa_force)
|
||||||
if not ok and is_cancelled():
|
if not ok and is_cancelled():
|
||||||
_blocks_errors[topic] = "Cancelled — progress is preserved"
|
_blocks_errors[topic] = "Cancelled — progress is preserved"
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|||||||
@@ -10,20 +10,24 @@ The heavy lifting is the existing per-block functions in blocks.py — each card
|
|||||||
work subdirectory + facts/artefakte paths, so their slot files never collide across blocks."""
|
work subdirectory + facts/artefakte paths, so their slot files never collide across blocks."""
|
||||||
|
|
||||||
import asyncio
|
import asyncio
|
||||||
|
import hashlib
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import re
|
import re
|
||||||
|
|
||||||
import database as db
|
import database as db
|
||||||
import blocks
|
import blocks
|
||||||
|
import embedding
|
||||||
from blocks import (
|
from blocks import (
|
||||||
ARTEFACT_TYPES, _artefacts_block, _facts_block, _levels_block, _match_sub,
|
ARTEFACT_TYPES, _artefacts_block, _facts_block, _konsolidiere_subblocks, _levels_block,
|
||||||
_question_pattern_block, _relevance_block, _subblocks_block, _outline_block,
|
_luecken_runde, _match_sub, _neg_set, _question_pattern_block, _relevance_block,
|
||||||
|
_sink_json, _subblocks_block, _outline_block,
|
||||||
)
|
)
|
||||||
|
from config import EMBEDDING_AKTIV, SUB_DUP_KANDIDAT_COS
|
||||||
from fsutil import atomic_write_json
|
from fsutil import atomic_write_json
|
||||||
from jsonio import read_json_file as _json_file
|
from jsonio import read_json_file as _json_file
|
||||||
from kanban import Flow, Stage
|
from kanban import Flow, Stage
|
||||||
from pipeline import GenContext, _log
|
from pipeline import FAILED, GenContext, _extra, _log, _prompt, _timeout, run_single_slot
|
||||||
from textkit import _norm_title, _title
|
from textkit import _norm_title, _title
|
||||||
|
|
||||||
log = logging.getLogger("creator.board_artefacts")
|
log = logging.getLogger("creator.board_artefacts")
|
||||||
@@ -40,6 +44,23 @@ def _safe(norm: str) -> str:
|
|||||||
return re.sub(r"\W+", "-", norm).strip("-")[:24] or "block"
|
return re.sub(r"\W+", "-", norm).strip("-")[:24] or "block"
|
||||||
|
|
||||||
|
|
||||||
|
def _sub_key(existing: set[str], sn: str) -> str:
|
||||||
|
"""Agents echo the short sub title while the sub row is keyed 'kurztitel: beschreibung' —
|
||||||
|
resolve to the stored key: exact, unambiguous prefix, then unambiguous substring
|
||||||
|
containment either way (agents paraphrase/truncate, measured 23 orphans of ~560 rows).
|
||||||
|
Ambiguous or unresolvable echoes stay unchanged (visible as QA orphan)."""
|
||||||
|
if sn in existing:
|
||||||
|
return sn
|
||||||
|
hits = [s for s in existing if s.startswith(sn + ":")]
|
||||||
|
if len(hits) == 1:
|
||||||
|
return hits[0]
|
||||||
|
if not hits:
|
||||||
|
hits = [s for s in sorted(existing) if sn in s or s in sn]
|
||||||
|
if len(hits) == 1:
|
||||||
|
return hits[0]
|
||||||
|
return sn
|
||||||
|
|
||||||
|
|
||||||
def _card_set_p(flow: Flow, norm: str):
|
def _card_set_p(flow: Flow, norm: str):
|
||||||
"""Per-card progress: the inner step messages land in-memory on the flow —
|
"""Per-card progress: the inner step messages land in-memory on the flow —
|
||||||
board_snapshot shows them as the card's info line + phase stepper while active.
|
board_snapshot shows them as the card's info line + phase stepper while active.
|
||||||
@@ -162,7 +183,9 @@ async def _proc_subblocks(ctx: GenContext, flow: Flow, files: dict, instructions
|
|||||||
sd = [s for s in seeds.get(norm, []) if s]
|
sd = [s for s in seeds.get(norm, []) if s]
|
||||||
if sd:
|
if sd:
|
||||||
instr = (instructions + "\n\nBereits identifizierte Unterpunkt-Kandidaten dieses "
|
instr = (instructions + "\n\nBereits identifizierte Unterpunkt-Kandidaten dieses "
|
||||||
"Blocks (unbedingt prüfen und, wenn belegt, aufnehmen):\n"
|
"Blocks (prüfen; wenn belegt UND noch nicht durch einen anderen Eintrag "
|
||||||
|
"abgedeckt, aufnehmen — nicht wörtlich übernehmen, sondern als eigenständige "
|
||||||
|
"Aussage formulieren):\n"
|
||||||
+ "\n".join(f"- {s}" for s in sd))
|
+ "\n".join(f"- {s}" for s in sd))
|
||||||
raw = await _subblocks_block(ctx, _card_set_p(flow, norm), _pfiles(files, norm),
|
raw = await _subblocks_block(ctx, _card_set_p(flow, norm), _pfiles(files, norm),
|
||||||
{1: _entry_line(p)}, instr, wipe=False, ns=f"{_safe(norm)}-",
|
{1: _entry_line(p)}, instr, wipe=False, ns=f"{_safe(norm)}-",
|
||||||
@@ -197,6 +220,29 @@ async def _proc_facts(ctx: GenContext, flow: Flow, files: dict, q: dict, folder,
|
|||||||
if bt in raw:
|
if bt in raw:
|
||||||
raw[bt] = [s for s in raw[bt] if _norm_title(s) not in sns]
|
raw[bt] = [s for s in raw[bt] if _norm_title(s) not in sns]
|
||||||
raw = {bt: subs for bt, subs in raw.items() if subs}
|
raw = {bt: subs for bt, subs in raw.items() if subs}
|
||||||
|
# In-block consolidation: two-judge panel folds same-statement/subset subs, bundles
|
||||||
|
# catalogs, drops off-topic ones — the facts are in hand (key points as evidence),
|
||||||
|
# questions/artefacts not yet built. Reported gaps get ONE follow-up finder round.
|
||||||
|
luecken = await _konsolidiere_subblocks(ctx, _pfiles(files, norm), raw, facts_map,
|
||||||
|
instructions, ns=f"{_safe(norm)}-",
|
||||||
|
lbl=f"{p.get('title', norm)} · ")
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return None
|
||||||
|
nachgefasst = 0
|
||||||
|
for bt, lk in (luecken or {}).items():
|
||||||
|
nachgefasst += await _luecken_runde(ctx, _pfiles(files, norm), bt, lk, raw, facts_map,
|
||||||
|
q, folder, instructions, ns=f"{_safe(norm)}-",
|
||||||
|
lbl=f"{p.get('title', norm)} · ", sources=p.get("sources"))
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return None
|
||||||
|
if nachgefasst: # close the loop: follow-up finds get the SAME duplicate test as the
|
||||||
|
# rest (new subs-hash → fresh judge files); their gap report is deliberately ignored
|
||||||
|
await _konsolidiere_subblocks(ctx, _pfiles(files, norm), raw, facts_map,
|
||||||
|
instructions, ns=f"{_safe(norm)}-",
|
||||||
|
lbl=f"{p.get('title', norm)} · ")
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return None
|
||||||
|
raw = {bt: subs for bt, subs in raw.items() if subs}
|
||||||
p["raw"], p["facts"] = raw, facts_map
|
p["raw"], p["facts"] = raw, facts_map
|
||||||
await db.kanban_set_payload(topic, BOARD, norm, p)
|
await db.kanban_set_payload(topic, BOARD, norm, p)
|
||||||
await db.kanban_advance(topic, BOARD, norm, "levels")
|
await db.kanban_advance(topic, BOARD, norm, "levels")
|
||||||
@@ -204,6 +250,174 @@ async def _proc_facts(ctx: GenContext, flow: Flow, files: dict, q: dict, folder,
|
|||||||
await _gather_cards(ctx, flow, cards, one)
|
await _gather_cards(ctx, flow, cards, one)
|
||||||
|
|
||||||
|
|
||||||
|
def _cross_schema(data) -> dict[int, str] | None:
|
||||||
|
"""{"pairs": {"1": "a"|"b"|"nein"}} → {pair_nr: verdict} · otherwise None."""
|
||||||
|
if not isinstance(data, dict) or not isinstance(data.get("pairs"), dict):
|
||||||
|
return None
|
||||||
|
out: dict[int, str] = {}
|
||||||
|
for k, v in data["pairs"].items():
|
||||||
|
try:
|
||||||
|
nr = int(k)
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
continue
|
||||||
|
s = str(v).strip().casefold()
|
||||||
|
if s in ("a", "b", "nein"):
|
||||||
|
out[nr] = s
|
||||||
|
return out or None
|
||||||
|
|
||||||
|
|
||||||
|
async def _proc_konsolidierung(ctx: GenContext, flow: Flow, files: dict, instructions: str, cards):
|
||||||
|
"""BARRIER/drain — cross-block sub dedup: the SAME statement carried by two blocks
|
||||||
|
(measured on Markdown: tab handling in 3 blocks, HTML blocks, backslash escapes — the
|
||||||
|
in-block paths never see these). Embedding candidates (block≠block, cos ≥
|
||||||
|
SUB_DUP_KANDIDAT_COS) go to a two-judge panel; UNANIMITY decides which block keeps the
|
||||||
|
statement. The loser leaves its card's raw/facts and turns DB `variant` — before
|
||||||
|
questions/artefacts exist, so no orphans. Fail-open on judge failure/dissent."""
|
||||||
|
topic = flow.topic
|
||||||
|
work_dir = flow.work_dir
|
||||||
|
package_norms = {c["card_id"] for c in cards}
|
||||||
|
entries: list[tuple[int, str, str]] = [] # (card idx, block title, sub title); idx -1 = context
|
||||||
|
for ci, c in enumerate(cards):
|
||||||
|
for bt, subs in (c["payload"].get("raw") or {}).items():
|
||||||
|
for s in subs:
|
||||||
|
entries.append((ci, bt, s))
|
||||||
|
n_pkg = len(entries)
|
||||||
|
# Context: consensus subs of blocks already PAST this barrier (late spawns via the
|
||||||
|
# gap-check feedback would otherwise never be compared). Context never folds —
|
||||||
|
# its card payload lives downstream (board-1 rule: confirmed context always wins).
|
||||||
|
for r in await db.list_subblocks(topic):
|
||||||
|
if r["status"] == "consensus" and r["block_norm"] not in package_norms:
|
||||||
|
entries.append((-1, r["block"], r["sub_title"]))
|
||||||
|
ctx_facts: dict[str, dict] = {} # facts of downstream cards (DB rows carry none yet)
|
||||||
|
for bc in await db.kanban_cards(topic, board=BOARD, kind="ablock"):
|
||||||
|
if bc["card_id"] not in package_norms:
|
||||||
|
for bt, fm in (bc["payload"].get("facts") or {}).items():
|
||||||
|
ctx_facts[_norm_title(bt)] = fm
|
||||||
|
|
||||||
|
async def _advance_all():
|
||||||
|
await db.kanban_advance_many(topic, BOARD, [(c["card_id"], "question_pattern") for c in cards])
|
||||||
|
flow.wake.set()
|
||||||
|
|
||||||
|
if n_pkg < 1 or len(entries) < 2 or not EMBEDDING_AKTIV or not await asyncio.to_thread(embedding.available):
|
||||||
|
await _advance_all()
|
||||||
|
return
|
||||||
|
sims = await asyncio.to_thread(embedding.embed_sims, [s for _, _, s in entries])
|
||||||
|
if sims is None:
|
||||||
|
await _advance_all()
|
||||||
|
return
|
||||||
|
negs = [_neg_set(s) for _, _, s in entries]
|
||||||
|
pairs = [(i, j) for i in range(len(entries)) for j in range(i + 1, len(entries))
|
||||||
|
if entries[i][0] != entries[j][0] and negs[i] == negs[j]
|
||||||
|
and float(sims[i][j]) >= SUB_DUP_KANDIDAT_COS]
|
||||||
|
if not pairs:
|
||||||
|
await _advance_all()
|
||||||
|
return
|
||||||
|
|
||||||
|
def _kp(ci: int, bt: str, s: str) -> list:
|
||||||
|
if ci < 0:
|
||||||
|
f = ctx_facts.get(_norm_title(bt)) or {}
|
||||||
|
else:
|
||||||
|
f = (cards[ci]["payload"].get("facts") or {}).get(bt) or {}
|
||||||
|
return (f.get(_norm_title(s)) or {}).get("key_points") or []
|
||||||
|
|
||||||
|
def _side(tag: str, ci: int, bt: str, s: str) -> str:
|
||||||
|
return f"{tag}: [Block: {bt}] {s}" + "".join(f"\n - {p}" for p in _kp(ci, bt, s))
|
||||||
|
|
||||||
|
lines = "\n\n".join(
|
||||||
|
f"{k}.\n{_side('A', *entries[i])}\n{_side('B', *entries[j])}"
|
||||||
|
for k, (i, j) in enumerate(pairs, 1))
|
||||||
|
h = hashlib.md5(lines.encode()).hexdigest()[:8]
|
||||||
|
paths = [work_dir / f"sub-crossblock-{h}-j{j}.json" for j in (1, 2)]
|
||||||
|
|
||||||
|
async def _judge(j, path):
|
||||||
|
if _cross_schema(_json_file(path)) is not None:
|
||||||
|
return # resume
|
||||||
|
status, _v = await run_single_slot(
|
||||||
|
ctx, f"Sub-Crossblock j{j}", key=f"blocks-{topic}-sub-crossblock-{h}-j{j}",
|
||||||
|
prompt=_prompt("Subblock-Crossblock", topic=topic, pairs=lines, extra=_extra(instructions)),
|
||||||
|
role="judge", capabilities="none",
|
||||||
|
payload=lambda result, p=path: _sink_json(result, p, _cross_schema),
|
||||||
|
timeout=_timeout("subblock_check", len(pairs)))
|
||||||
|
if status == FAILED:
|
||||||
|
_log(topic, f"Sub-Crossblock j{j} ohne Ergebnis — fail-open")
|
||||||
|
|
||||||
|
await asyncio.gather(*[_judge(j, p) for j, p in zip((1, 2), paths)])
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return
|
||||||
|
outs = [o for p in paths if (o := _cross_schema(_json_file(p))) is not None]
|
||||||
|
if len(outs) == 1: # Ersatz-Richter statt fail-open bei EINEM Ausfall
|
||||||
|
ersatz = work_dir / f"sub-crossblock-{h}-jE.json"
|
||||||
|
await _judge("E", ersatz)
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return
|
||||||
|
outs = [o for p in [*paths, ersatz] if (o := _cross_schema(_json_file(p))) is not None]
|
||||||
|
journal = {"paare": len(pairs), "richter": len(outs), "gefaltet": [], "verdicts": []}
|
||||||
|
gone: set[int] = set()
|
||||||
|
touched: set[int] = set()
|
||||||
|
if len(outs) == 2:
|
||||||
|
final = {k: (outs[0].get(k, "nein") if outs[0].get(k, "nein") == outs[1].get(k, "nein")
|
||||||
|
else "uneinig") for k in range(1, len(pairs) + 1)}
|
||||||
|
disputed = [k for k, v in final.items() if v == "uneinig"]
|
||||||
|
if disputed: # tie-breaker: a third judge sees ONLY the disputed pairs, majority 2/3
|
||||||
|
d_lines = "\n\n".join(
|
||||||
|
f"{x}.\n{_side('A', *entries[pairs[k - 1][0]])}\n{_side('B', *entries[pairs[k - 1][1]])}"
|
||||||
|
for x, k in enumerate(disputed, 1))
|
||||||
|
p3 = work_dir / f"sub-crossblock-{h}-j3.json"
|
||||||
|
if _cross_schema(_json_file(p3)) is None:
|
||||||
|
status, _v = await run_single_slot(
|
||||||
|
ctx, "Sub-Crossblock j3", key=f"blocks-{topic}-sub-crossblock-{h}-j3",
|
||||||
|
prompt=_prompt("Subblock-Crossblock", topic=topic, pairs=d_lines, extra=_extra(instructions)),
|
||||||
|
role="judge", capabilities="none",
|
||||||
|
payload=lambda result, p=p3: _sink_json(result, p, _cross_schema),
|
||||||
|
timeout=_timeout("subblock_check", len(disputed)))
|
||||||
|
if status == FAILED:
|
||||||
|
_log(topic, "Sub-Crossblock j3 ohne Ergebnis — strittige Paare bleiben")
|
||||||
|
if ctx.is_cancelled():
|
||||||
|
return
|
||||||
|
v3 = _cross_schema(_json_file(p3)) or {}
|
||||||
|
for x, k in enumerate(disputed, 1):
|
||||||
|
t = v3.get(x, "nein")
|
||||||
|
if t in (outs[0].get(k, "nein"), outs[1].get(k, "nein")):
|
||||||
|
final[k] = t # majority 2/3; anything else stays disputed → no fold
|
||||||
|
for k, (i, j) in enumerate(pairs, 1):
|
||||||
|
verdict = final[k]
|
||||||
|
journal["verdicts"].append({"a": f"{entries[i][1]} · {entries[i][2]}",
|
||||||
|
"b": f"{entries[j][1]} · {entries[j][2]}",
|
||||||
|
"verdict": verdict})
|
||||||
|
if verdict not in ("a", "b"):
|
||||||
|
continue
|
||||||
|
lose = j if verdict == "a" else i
|
||||||
|
if entries[lose][0] < 0: # context never folds — the package side goes instead
|
||||||
|
lose = i if lose == j else j
|
||||||
|
keep = i if lose == j else j
|
||||||
|
if lose in gone or keep in gone: # keeper already folded → don't chain away the content
|
||||||
|
continue
|
||||||
|
ci, bt, s = entries[lose]
|
||||||
|
p = cards[ci]["payload"]
|
||||||
|
if s in (p.get("raw") or {}).get(bt, []):
|
||||||
|
p["raw"][bt].remove(s)
|
||||||
|
(p.get("facts") or {}).get(bt, {}).pop(_norm_title(s), None)
|
||||||
|
sc = (p.get("sidecar") or {}).get(bt)
|
||||||
|
if isinstance(sc, list): # questions/artefacts consume the sidecar downstream
|
||||||
|
p["sidecar"][bt] = [e for e in sc
|
||||||
|
if _norm_title(str((e or {}).get("title", ""))) != _norm_title(s)]
|
||||||
|
await db.set_subblock_fields(topic, _norm_title(bt), _norm_title(s), status="variant")
|
||||||
|
gone.add(lose)
|
||||||
|
touched.add(ci)
|
||||||
|
journal["gefaltet"].append({"weg": f"{bt} · {s}",
|
||||||
|
"bleibt": f"{entries[keep][1]} · {entries[keep][2]}"})
|
||||||
|
elif outs:
|
||||||
|
_log(topic, "Sub-Crossblock: nur 1/2 Richter — fail-open")
|
||||||
|
for ci in touched:
|
||||||
|
p = cards[ci]["payload"]
|
||||||
|
p["raw"] = {bt: subs for bt, subs in (p.get("raw") or {}).items() if subs}
|
||||||
|
await db.kanban_set_payload(topic, BOARD, cards[ci]["card_id"], p)
|
||||||
|
if journal["gefaltet"]:
|
||||||
|
_log(topic, f"Sub-Crossblock: {len(journal['gefaltet'])} blockübergreifende Dublette(n) gefaltet")
|
||||||
|
atomic_write_json(work_dir / f"sub-crossblock-{h}.json", journal, indent=1)
|
||||||
|
await _advance_all()
|
||||||
|
|
||||||
|
|
||||||
async def _proc_levels(ctx: GenContext, flow: Flow, files: dict, instructions: str, cards):
|
async def _proc_levels(ctx: GenContext, flow: Flow, files: dict, instructions: str, cards):
|
||||||
topic = flow.topic
|
topic = flow.topic
|
||||||
|
|
||||||
@@ -247,7 +461,7 @@ async def _proc_relevance(ctx: GenContext, flow: Flow, files: dict, instructions
|
|||||||
sub["relevance"] = rel.get(gid, "relevant")
|
sub["relevance"] = rel.get(gid, "relevant")
|
||||||
p["sidecar"] = sidecar
|
p["sidecar"] = sidecar
|
||||||
await db.kanban_set_payload(topic, BOARD, norm, p)
|
await db.kanban_set_payload(topic, BOARD, norm, p)
|
||||||
await db.kanban_advance(topic, BOARD, norm, "question_pattern")
|
await db.kanban_advance(topic, BOARD, norm, "konsolidierung")
|
||||||
|
|
||||||
await _gather_cards(ctx, flow, cards, one)
|
await _gather_cards(ctx, flow, cards, one)
|
||||||
|
|
||||||
@@ -320,6 +534,18 @@ async def _proc_finalize(ctx: GenContext, flow: Flow, files: dict, cards):
|
|||||||
atomic_write_json(files["artefakte"], art_global, indent=1)
|
atomic_write_json(files["artefakte"], art_global, indent=1)
|
||||||
# DB mirrors — per block only (no global deletes)
|
# DB mirrors — per block only (no global deletes)
|
||||||
await blocks._mirror_sidecar_db(topic, sidecar)
|
await blocks._mirror_sidecar_db(topic, sidecar)
|
||||||
|
# stale question/artefact rows of a PREVIOUS run keyed to gone subs: finalize only
|
||||||
|
# upserts, so re-runs left orphans (measured: 28). subblocks rows stay — QA needs
|
||||||
|
# the variant/discarded statuses, and the sidecar mirror re-writes only consensus.
|
||||||
|
await db.delete_question_pattern(topic, _norm_title(title))
|
||||||
|
await db.delete_sub_artefakte(topic, _norm_title(title))
|
||||||
|
sub_keys: dict[str, set[str]] = {}
|
||||||
|
|
||||||
|
async def _keys(bnorm: str) -> set[str]:
|
||||||
|
if bnorm not in sub_keys:
|
||||||
|
sub_keys[bnorm] = {r["sub_norm"] for r in await db.list_subblocks(topic, bnorm)}
|
||||||
|
return sub_keys[bnorm]
|
||||||
|
|
||||||
for btitle, entries in pattern.items():
|
for btitle, entries in pattern.items():
|
||||||
bnorm = _norm_title(btitle)
|
bnorm = _norm_title(btitle)
|
||||||
for e in entries if isinstance(entries, list) else []:
|
for e in entries if isinstance(entries, list) else []:
|
||||||
@@ -327,6 +553,7 @@ async def _proc_finalize(ctx: GenContext, flow: Flow, files: dict, cards):
|
|||||||
sn = _norm_title(sub)
|
sn = _norm_title(sub)
|
||||||
question = str(e.get("question", "")).strip()
|
question = str(e.get("question", "")).strip()
|
||||||
if bnorm and sn and question:
|
if bnorm and sn and question:
|
||||||
|
sn = _sub_key(await _keys(bnorm), sn)
|
||||||
await db.upsert_question_pattern(topic, bnorm, sn, btitle, sub, question)
|
await db.upsert_question_pattern(topic, bnorm, sn, btitle, sub, question)
|
||||||
btitles = list(sidecar.keys())
|
btitles = list(sidecar.keys())
|
||||||
for typ in ARTEFACT_TYPES:
|
for typ in ARTEFACT_TYPES:
|
||||||
@@ -335,6 +562,7 @@ async def _proc_finalize(ctx: GenContext, flow: Flow, files: dict, cards):
|
|||||||
bnorm, sn = _norm_title(bt), _norm_title(str(e.get("subblock", "")))
|
bnorm, sn = _norm_title(bt), _norm_title(str(e.get("subblock", "")))
|
||||||
if not bnorm or not sn:
|
if not bnorm or not sn:
|
||||||
continue
|
continue
|
||||||
|
sn = _sub_key(await _keys(bnorm), sn)
|
||||||
data = json.dumps({k: v for k, v in e.items() if k not in ("block", "subblock")},
|
data = json.dumps({k: v for k, v in e.items() if k not in ("block", "subblock")},
|
||||||
ensure_ascii=False)
|
ensure_ascii=False)
|
||||||
await db.put_sub_artifact(topic, bnorm, sn, typ, data, bt, str(e.get("subblock", "")))
|
await db.put_sub_artifact(topic, bnorm, sn, typ, data, bt, str(e.get("subblock", "")))
|
||||||
@@ -391,6 +619,11 @@ def artefact_stages(ctx: GenContext, flow: Flow, files: dict, q: dict, folder,
|
|||||||
Stage(BOARD, "facts", lambda cs: _proc_facts(ctx, flow, files, q, folder, instructions, cs)),
|
Stage(BOARD, "facts", lambda cs: _proc_facts(ctx, flow, files, q, folder, instructions, cs)),
|
||||||
Stage(BOARD, "levels", lambda cs: _proc_levels(ctx, flow, files, instructions, cs)),
|
Stage(BOARD, "levels", lambda cs: _proc_levels(ctx, flow, files, instructions, cs)),
|
||||||
Stage(BOARD, "relevance", lambda cs: _proc_relevance(ctx, flow, files, instructions, cs)),
|
Stage(BOARD, "relevance", lambda cs: _proc_relevance(ctx, flow, files, instructions, cs)),
|
||||||
|
# Barrier sits AFTER the sub-local stages: cards used to idle here median 36 min
|
||||||
|
# while levels/relevance work was still ahead of them
|
||||||
|
Stage(BOARD, "konsolidierung",
|
||||||
|
lambda cs: _proc_konsolidierung(ctx, flow, files, instructions, cs),
|
||||||
|
barrier=True, drain=True),
|
||||||
Stage(BOARD, "question_pattern",
|
Stage(BOARD, "question_pattern",
|
||||||
lambda cs: _proc_question_pattern(ctx, flow, files, instructions, cs)),
|
lambda cs: _proc_question_pattern(ctx, flow, files, instructions, cs)),
|
||||||
Stage(BOARD, "artefacts", lambda cs: _proc_artefacts(ctx, flow, files, instructions, cs)),
|
Stage(BOARD, "artefacts", lambda cs: _proc_artefacts(ctx, flow, files, instructions, cs)),
|
||||||
|
|||||||
@@ -27,7 +27,10 @@ import hashlib
|
|||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import math
|
import math
|
||||||
|
import re
|
||||||
|
import unicodedata
|
||||||
import uuid
|
import uuid
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
import database as db
|
import database as db
|
||||||
import embedding
|
import embedding
|
||||||
@@ -43,9 +46,10 @@ from blocks import (
|
|||||||
_filter_schema, _filter_suspect, _is_artifact, _is_named_statement,
|
_filter_schema, _filter_suspect, _is_artifact, _is_named_statement,
|
||||||
_is_parentless_noise, _is_reference, _pairs_schema, _read,
|
_is_parentless_noise, _is_reference, _pairs_schema, _read,
|
||||||
_relation_conflict, _root, _supplement_schema, _text_sections, _umbrella_schema,
|
_relation_conflict, _root, _supplement_schema, _text_sections, _umbrella_schema,
|
||||||
_aspect_marker, _title_variants, _evidence_pack, _sink_json, source_folder,
|
_aspect_marker, _title_variants, _corpus_files, _evidence_pack, _sink_json, source_folder,
|
||||||
)
|
)
|
||||||
from config import (
|
from config import (QA_GATE_NOTE, QA_GATE_LLM,
|
||||||
|
|
||||||
BLOCKS_GRUPPIERUNG_AKTIV, EMBEDDING_AKTIV, EMBEDDING_BLOCK_CAP,
|
BLOCKS_GRUPPIERUNG_AKTIV, EMBEDDING_AKTIV, EMBEDDING_BLOCK_CAP,
|
||||||
EMBEDDING_SIBLING_CAP, EMBEDDING_SIBLING_FLOOR, FRAGMENT_MIN_COS,
|
EMBEDDING_SIBLING_CAP, EMBEDDING_SIBLING_FLOOR, FRAGMENT_MIN_COS,
|
||||||
GROUP_MIN_COS_FLOOR, GROUP_RECONCILE_FLOOR,
|
GROUP_MIN_COS_FLOOR, GROUP_RECONCILE_FLOOR,
|
||||||
@@ -410,11 +414,82 @@ def _rep(rows: list[dict]) -> dict:
|
|||||||
return _canonical(cands, list(range(len(cands))), set())
|
return _canonical(cands, list(range(len(cands))), set())
|
||||||
|
|
||||||
|
|
||||||
|
_ANKER_STOP = {"der", "die", "das", "und", "oder", "für", "mit", "von", "des", "den", "dem",
|
||||||
|
"ein", "eine", "einer", "the", "and", "for", "problem", "probleme",
|
||||||
|
"algorithmus", "algorithmen", "definition", "satz", "lemma", "beispiel",
|
||||||
|
"methode", "verfahren"}
|
||||||
|
|
||||||
|
|
||||||
|
def _korpus_tokens(folder) -> set[str]:
|
||||||
|
"""All corpus word tokens (≥3 chars, casefolded) — anchor base for the reader-title gate."""
|
||||||
|
toks: set[str] = set()
|
||||||
|
for f in _corpus_files(folder, None):
|
||||||
|
try:
|
||||||
|
toks |= set(re.findall(r"\w{3,}", f.read_text(encoding="utf-8").casefold()))
|
||||||
|
except OSError:
|
||||||
|
continue
|
||||||
|
return toks
|
||||||
|
|
||||||
|
|
||||||
|
def _hat_anker(title: str, ctoks: set[str]) -> bool:
|
||||||
|
"""≥1 distinctive title token appears in the corpus — digit-suffix tolerant:
|
||||||
|
'∆TSP1' → 'tsp1' → 'tsp' (the corpus tokenizes '∆TSP' to 'tsp')."""
|
||||||
|
for t in re.findall(r"\w{3,}", title.casefold()):
|
||||||
|
if t in _ANKER_STOP:
|
||||||
|
continue
|
||||||
|
forms = {t}
|
||||||
|
a = unicodedata.normalize("NFKD", t).encode("ascii", "ignore").decode()
|
||||||
|
if len(a) >= 3:
|
||||||
|
forms.add(a) # Symbol-Präfixe (δtsp1 → tsp1) — der Korpus-Tokenizer kennt kein ∆
|
||||||
|
forms |= {f2 for f in list(forms) if len(f2 := f.rstrip("0123456789")) >= 3}
|
||||||
|
if any(ct == f or ct.startswith(f) for f in forms for ct in ctoks):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
async def _anker_beleg(ctx: GenContext, flow: Flow, kandidaten: list[tuple[str, dict]]) -> set[str]:
|
||||||
|
"""Evidence judge for quorum titles WITHOUT any corpus anchor — two readers naming the
|
||||||
|
same famous canon independently beat the quorum although the material never mentions it
|
||||||
|
(measured: 'Königsberger Brückenproblem', 0 corpus hits). FAIL-OPEN: the titles carry
|
||||||
|
2-reader backing, only an explicit 'nein' rejects. → card_ids to reject."""
|
||||||
|
topic = flow.topic
|
||||||
|
folder = source_folder(topic)
|
||||||
|
lines = []
|
||||||
|
for k, (cid, p) in enumerate(kandidaten, 1):
|
||||||
|
ev = _evidence_pack(folder, None, [p.get("title", ""), p.get("description") or ""], budget=4000)
|
||||||
|
lines.append(f"{k}. {p.get('title', '')} — {p.get('description') or ''}\nAUSZÜGE:\n"
|
||||||
|
f"{ev or '(keine passenden Auszüge im Material gefunden)'}")
|
||||||
|
h = _h(*[cid for cid, _ in kandidaten])
|
||||||
|
path = flow.work_dir / f"anker-beleg-{h}.json"
|
||||||
|
ids = set(range(1, len(kandidaten) + 1))
|
||||||
|
verdict = _yesno_schema(_json_file(path), ids)
|
||||||
|
if verdict is None:
|
||||||
|
status, verdict = await run_single_slot(
|
||||||
|
ctx, "Anker-Beleg", key=f"blocks-{topic}-anker-beleg-{h}",
|
||||||
|
prompt=_prompt("Blocks-Supplement-Beleg", topic=topic, proposals="\n\n".join(lines),
|
||||||
|
extra=_extra(flow.state.get("instructions", ""))),
|
||||||
|
role="judge", capabilities="none",
|
||||||
|
payload=lambda result, p2=path, i=ids: _sink_json(result, p2, lambda d: _yesno_schema(d, i)),
|
||||||
|
timeout=_timeout("selection_mapping", len(kandidaten)))
|
||||||
|
if status != OK or not isinstance(verdict, dict):
|
||||||
|
verdict = {} # fail-open
|
||||||
|
return {cid for k, (cid, _p) in enumerate(kandidaten, 1) if str(verdict.get(k, "ja")) == "nein"}
|
||||||
|
|
||||||
|
|
||||||
async def _proc_consensus_gate(ctx: GenContext, flow: Flow, cards):
|
async def _proc_consensus_gate(ctx: GenContext, flow: Flow, cards):
|
||||||
"""Code gate: reader union ≥2 (or supplement) passes; single finds → clarify.
|
"""Code gate: reader union ≥2 (or supplement) passes; single finds → clarify.
|
||||||
Reference-titled consensus clusters also go to clarify (majority quorum + rename)."""
|
Reference-titled consensus clusters also go to clarify (majority quorum + rename).
|
||||||
|
uni/projekt: quorum titles without ANY corpus anchor face the evidence judge —
|
||||||
|
reader co-hallucination of famous canon beats the quorum otherwise."""
|
||||||
topic = flow.topic
|
topic = flow.topic
|
||||||
moves = []
|
moves = []
|
||||||
|
anker_kandidaten: list[tuple[str, dict]] = []
|
||||||
|
folder = source_folder(topic)
|
||||||
|
ctoks = None
|
||||||
|
if folder is not None:
|
||||||
|
ctoks = flow.state.get("korpus_tokens")
|
||||||
|
if ctoks is None:
|
||||||
|
ctoks = flow.state["korpus_tokens"] = await asyncio.to_thread(_korpus_tokens, folder)
|
||||||
for c in cards:
|
for c in cards:
|
||||||
cid = c["card_id"]
|
cid = c["card_id"]
|
||||||
rows = await _member_rows(topic, cid)
|
rows = await _member_rows(topic, cid)
|
||||||
@@ -431,11 +506,24 @@ async def _proc_consensus_gate(ctx: GenContext, flow: Flow, cards):
|
|||||||
if _is_reference(rep["title"]) and not supplement:
|
if _is_reference(rep["title"]) and not supplement:
|
||||||
p["quorum"] = "majority" # consensus reference title: rename/exam, not the hard bar
|
p["quorum"] = "majority" # consensus reference title: rename/exam, not the hard bar
|
||||||
moves.append((cid, "clarify"))
|
moves.append((cid, "clarify"))
|
||||||
|
elif ctoks and not supplement and not _hat_anker(rep["title"], ctoks):
|
||||||
|
anker_kandidaten.append((cid, p))
|
||||||
else:
|
else:
|
||||||
moves.append((cid, "naming"))
|
moves.append((cid, "naming"))
|
||||||
else:
|
else:
|
||||||
moves.append((cid, "clarify"))
|
moves.append((cid, "clarify"))
|
||||||
await db.kanban_set_payload(topic, BOARD, cid, p)
|
await db.kanban_set_payload(topic, BOARD, cid, p)
|
||||||
|
if anker_kandidaten:
|
||||||
|
weg = await _anker_beleg(ctx, flow, anker_kandidaten)
|
||||||
|
for cid, p in anker_kandidaten:
|
||||||
|
if cid in weg:
|
||||||
|
p["reason"] = "kein-beleg"
|
||||||
|
await db.kanban_set_payload(topic, BOARD, cid, p)
|
||||||
|
moves.append((cid, "rejected"))
|
||||||
|
else:
|
||||||
|
moves.append((cid, "naming"))
|
||||||
|
if weg:
|
||||||
|
_log(topic, f"Anker-Beleg: {len(weg)} Quorum-Titel ohne Materialbeleg verworfen")
|
||||||
await db.kanban_advance_many(topic, BOARD, moves)
|
await db.kanban_advance_many(topic, BOARD, moves)
|
||||||
flow.wake.set()
|
flow.wake.set()
|
||||||
|
|
||||||
@@ -1268,12 +1356,18 @@ async def _supplement_producer(ctx: GenContext, flow: Flow, titles: list[str]):
|
|||||||
path = flow.work_dir / "supplement.json"
|
path = flow.work_dir / "supplement.json"
|
||||||
supplements = _supplement_schema(_json_file(path))
|
supplements = _supplement_schema(_json_file(path))
|
||||||
if supplements is None:
|
if supplements is None:
|
||||||
|
# Source-bound topics (uni/projekt/link): the MATERIAL defines the scope — the agent
|
||||||
|
# compares inventory vs. material (files, no web). Only pure "thema" topics research
|
||||||
|
# the canon on the web (measured: the web agent proposed 22 textbook blocks the
|
||||||
|
# script never treats, all discarded by the Beleg gate — 7 wasted minutes).
|
||||||
|
folder = source_folder(topic)
|
||||||
|
template, caps = ("Blocks-Supplement-Material", "files") if folder else ("Blocks-Supplement", "full")
|
||||||
status, supplements = await run_single_slot(
|
status, supplements = await run_single_slot(
|
||||||
ctx, "Supplement", key=f"blocks-{topic}-supplement",
|
ctx, "Supplement", key=f"blocks-{topic}-supplement",
|
||||||
prompt=_prompt("Blocks-Supplement", topic=topic,
|
prompt=_prompt(template, topic=topic, project=folder,
|
||||||
blocks="\n".join(f"- {t}" for t in titles),
|
blocks="\n".join(f"- {t}" for t in titles),
|
||||||
out_path=path, extra=_extra(flow.state.get("instructions", ""))),
|
out_path=path, extra=_extra(flow.state.get("instructions", ""))),
|
||||||
role="quick", capabilities="full",
|
role="quick", capabilities=caps,
|
||||||
payload=lambda result, p=path: _supplement_schema(_json_file(p)),
|
payload=lambda result, p=path: _supplement_schema(_json_file(p)),
|
||||||
timeout=_timeout("ergaenzung"))
|
timeout=_timeout("ergaenzung"))
|
||||||
if status == CANCELLED:
|
if status == CANCELLED:
|
||||||
@@ -1429,22 +1523,35 @@ async def _preload_state(flow: Flow):
|
|||||||
|
|
||||||
|
|
||||||
async def run_boards(ctx: GenContext, set_p, files: dict, q: dict, folder, instructions: str,
|
async def run_boards(ctx: GenContext, set_p, files: dict, q: dict, folder, instructions: str,
|
||||||
research: bool = True, artefacts: bool = True) -> bool:
|
research: bool = True, artefacts: bool = True, qa_force: bool = False) -> bool:
|
||||||
"""Run the inventory board (plus board 2 „Artefakte") until quiescence.
|
"""Run the inventory board (plus board 2 „Artefakte") until quiescence.
|
||||||
research=False = Continue: drain the existing queue, search nothing new."""
|
research=False = Continue: drain the existing queue, search nothing new.
|
||||||
|
qa_force=True overrides a failed QA gate (user clicked „Trotzdem fortsetzen")."""
|
||||||
topic = ctx.topic
|
topic = ctx.topic
|
||||||
flow = Flow(topic, files["arbeit"])
|
flow = Flow(topic, files["arbeit"])
|
||||||
flow.state["instructions"] = instructions
|
flow.state["instructions"] = instructions
|
||||||
|
flow.state["qa_force"] = qa_force
|
||||||
|
run_id = f"{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M')}-{uuid.uuid4().hex[:4]}"
|
||||||
|
flow.state["run_id"] = run_id
|
||||||
|
flow.state["run_started"] = datetime.now(timezone.utc).isoformat()
|
||||||
|
db.set_current_run(topic, run_id) # every event of this flow carries the run_id
|
||||||
if q["type"] == "link" and folder:
|
if q["type"] == "link" and folder:
|
||||||
pages = await db.list_content(topic)
|
pages = await db.list_content(topic)
|
||||||
flow.state["pages"] = pages or sorted(set(_crawl_index(folder).values()))
|
flow.state["pages"] = pages or sorted(set(_crawl_index(folder).values()))
|
||||||
await _preload_state(flow)
|
await _preload_state(flow)
|
||||||
stages = inventory_stages(ctx, flow)
|
stages = inventory_stages(ctx, flow)
|
||||||
|
inv_names = [st.stage for st in stages]
|
||||||
if artefacts:
|
if artefacts:
|
||||||
import board_artefacts # lazy — board_artefacts imports blocks too
|
import board_artefacts # lazy — board_artefacts imports blocks too
|
||||||
flow.state["spawn_artefact"] = board_artefacts.make_spawner(topic, files)
|
flow.state["spawn_artefact"] = board_artefacts.make_spawner(topic, files)
|
||||||
await board_artefacts.ensure_outline_card(topic)
|
await board_artefacts.ensure_outline_card(topic)
|
||||||
stages += board_artefacts.artefact_stages(ctx, flow, files, q, folder, instructions)
|
stages += board_artefacts.artefact_stages(ctx, flow, files, q, folder, instructions)
|
||||||
|
if QA_GATE_NOTE > 0:
|
||||||
|
# QA gate: board 2 waits until the inventory QA passed (or the user forces).
|
||||||
|
# Costs pipelining (board 2 no longer starts per finished block) but saves
|
||||||
|
# tokens on a bad foundation — the watcher below runs the QA and decides.
|
||||||
|
sub = next(st for st in stages if st.stage == "subblocks")
|
||||||
|
sub.gate = lambda: bool(flow.state.get("qa_ok") or flow.state.get("qa_force"))
|
||||||
stages = chain_stages(stages)
|
stages = chain_stages(stages)
|
||||||
if artefacts:
|
if artefacts:
|
||||||
# Outline needs every block's TITLE + FACTS, nothing later: cut the post-facts
|
# Outline needs every block's TITLE + FACTS, nothing later: cut the post-facts
|
||||||
@@ -1469,16 +1576,98 @@ async def run_boards(ctx: GenContext, set_p, files: dict, q: dict, folder, instr
|
|||||||
ctx, flow, q, folder, instructions, f"x{flow.next_tag()}")
|
ctx, flow, q, folder, instructions, f"x{flow.next_tag()}")
|
||||||
# cancel hook: blocks.cancel_blocks flips is_cancelled; stop the flow with it
|
# cancel hook: blocks.cancel_blocks flips is_cancelled; stop the flow with it
|
||||||
stopper = asyncio.create_task(_stop_on_cancel(ctx, flow))
|
stopper = asyncio.create_task(_stop_on_cancel(ctx, flow))
|
||||||
|
watcher = (asyncio.create_task(_qa_gate_watch(ctx, flow, inv_names, set_p))
|
||||||
|
if artefacts and QA_GATE_NOTE > 0 else None)
|
||||||
try:
|
try:
|
||||||
await kanban.run_flow(flow, stages, [_as_producer(p) for p in producers], set_p)
|
await kanban.run_flow(flow, stages, [_as_producer(p) for p in producers], set_p)
|
||||||
finally:
|
finally:
|
||||||
stopper.cancel()
|
stopper.cancel()
|
||||||
|
if watcher:
|
||||||
|
watcher.cancel()
|
||||||
|
db.set_current_run(topic, None)
|
||||||
if ctx.is_cancelled():
|
if ctx.is_cancelled():
|
||||||
return False
|
return False
|
||||||
|
if flow.state.get("qa_paused"):
|
||||||
|
return True # kein Fehler: Flow hielt am QA-Gate, Karten warten in subblocks
|
||||||
await _write_final(topic, files)
|
await _write_final(topic, files)
|
||||||
|
await _write_run_summary(topic, flow)
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
_QA_GATE_POLL = 2.0 # Sekunden zwischen Quiescence-Checks des QA-Wächters
|
||||||
|
|
||||||
|
|
||||||
|
async def _qa_gate_watch(ctx: GenContext, flow: Flow, inv_names: list[str], set_p):
|
||||||
|
"""Companion task: once the inventory is quiescent, run the QA once and decide —
|
||||||
|
open the board-2 gate or pause the flow. Fail-OPEN on errors (QA is a helper,
|
||||||
|
not a jailer); qa_force short-circuits to open."""
|
||||||
|
topic = flow.topic
|
||||||
|
try:
|
||||||
|
while not flow.stop:
|
||||||
|
if (flow.research_done and not flow.active_in(inv_names)
|
||||||
|
and await db.kanban_count(topic, inv_names) == 0):
|
||||||
|
break
|
||||||
|
await asyncio.sleep(_QA_GATE_POLL)
|
||||||
|
if flow.stop or ctx.is_cancelled():
|
||||||
|
return
|
||||||
|
if flow.state.get("qa_force"):
|
||||||
|
flow.state["qa_ok"] = True
|
||||||
|
flow.wake.set()
|
||||||
|
return
|
||||||
|
set_p("QA prüft das Inventar…")
|
||||||
|
import qa
|
||||||
|
report = await qa.qa_report(topic, llm=QA_GATE_LLM)
|
||||||
|
note = float(report["note"]) if report else 10.0
|
||||||
|
flow.state["qa_note"] = note
|
||||||
|
if report:
|
||||||
|
try: # Report-Persistenz ist Komfort — ein Schreibfehler darf das Gate nicht öffnen
|
||||||
|
await asyncio.to_thread(qa._write_report, report)
|
||||||
|
except Exception:
|
||||||
|
log.exception("[%s] QA-Report schreiben fehlgeschlagen", topic)
|
||||||
|
if note >= QA_GATE_NOTE:
|
||||||
|
_log(topic, f"QA-Gate: Note {note} ≥ {QA_GATE_NOTE} — Board 2 startet")
|
||||||
|
flow.state["qa_ok"] = True
|
||||||
|
else:
|
||||||
|
_log(topic, f"QA-Gate: Note {note} < {QA_GATE_NOTE} — pausiert (Continue erzwingt)")
|
||||||
|
set_p(f"QA-Note {note} < {QA_GATE_NOTE} — pausiert")
|
||||||
|
flow.state["qa_paused"] = True
|
||||||
|
flow.stop = True
|
||||||
|
flow.wake.set()
|
||||||
|
except asyncio.CancelledError:
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
log.exception("[%s] QA-Gate fehlgeschlagen — Gate offen (fail-open)", topic)
|
||||||
|
flow.state["qa_ok"] = True
|
||||||
|
flow.wake.set()
|
||||||
|
|
||||||
|
|
||||||
|
async def _write_run_summary(topic: str, flow: Flow):
|
||||||
|
"""lauf-summary.json: the per-run numbers block QA diffs against. Never fatal."""
|
||||||
|
try:
|
||||||
|
run_id = flow.state.get("run_id", "")
|
||||||
|
started = flow.state.get("run_started", "")
|
||||||
|
finished = datetime.now(timezone.utc).isoformat()
|
||||||
|
dauer = ""
|
||||||
|
if started:
|
||||||
|
dauer = round((datetime.fromisoformat(finished) - datetime.fromisoformat(started)).total_seconds() / 60, 1)
|
||||||
|
summary = {"run_id": run_id, "topic": topic, "started": started, "finished": finished,
|
||||||
|
"dauer_min": dauer, "boards": await db.kanban_stage_counts(topic),
|
||||||
|
**await db.events_run_summary(topic, run_id)}
|
||||||
|
try: # Abschluss-QA MIT Judges: sub_dubletten/unechte werden beurteilt — erst damit
|
||||||
|
import qa # ist note_artefakte belastbar (Kandidatenliste allein zählt nicht)
|
||||||
|
report = await qa.qa_report(topic, llm=True)
|
||||||
|
if report:
|
||||||
|
summary["note"] = report["note"]
|
||||||
|
summary["note_artefakte"] = report.get("note_artefakte")
|
||||||
|
summary["artefakte"] = report.get("artefakte", {})
|
||||||
|
await asyncio.to_thread(qa._write_report, report)
|
||||||
|
except Exception:
|
||||||
|
log.exception("[%s] Abschluss-QA fehlgeschlagen", topic)
|
||||||
|
atomic_write_json(flow.work_dir / "lauf-summary.json", summary, indent=1)
|
||||||
|
except Exception:
|
||||||
|
log.exception("[%s] lauf-summary fehlgeschlagen", topic)
|
||||||
|
|
||||||
|
|
||||||
async def _stop_on_cancel(ctx: GenContext, flow: Flow):
|
async def _stop_on_cancel(ctx: GenContext, flow: Flow):
|
||||||
while not flow.stop:
|
while not flow.stop:
|
||||||
if ctx.is_cancelled():
|
if ctx.is_cancelled():
|
||||||
@@ -1520,6 +1709,7 @@ COLUMNS = [
|
|||||||
("artefacts", "facts", "Fakten", "ablock"),
|
("artefacts", "facts", "Fakten", "ablock"),
|
||||||
("artefacts", "levels", "Stufen", "ablock"),
|
("artefacts", "levels", "Stufen", "ablock"),
|
||||||
("artefacts", "relevance", "Relevanz", "ablock"),
|
("artefacts", "relevance", "Relevanz", "ablock"),
|
||||||
|
("artefacts", "konsolidierung", "Konsolidierung", "ablock"),
|
||||||
("artefacts", "question_pattern", "Fragen", "ablock"),
|
("artefacts", "question_pattern", "Fragen", "ablock"),
|
||||||
("artefacts", "artefacts", "Lernkarten", "ablock"),
|
("artefacts", "artefacts", "Lernkarten", "ablock"),
|
||||||
("artefacts", "finalize", "Zusammenführen", "ablock"),
|
("artefacts", "finalize", "Zusammenführen", "ablock"),
|
||||||
@@ -1530,7 +1720,7 @@ COLUMNS = [
|
|||||||
_TITLE_STAGES = ["ingest", "cluster"]
|
_TITLE_STAGES = ["ingest", "cluster"]
|
||||||
_CLUSTER_STAGES = ["pair_check", "consensus_gate", "clarify", "naming", "naming_check"]
|
_CLUSTER_STAGES = ["pair_check", "consensus_gate", "clarify", "naming", "naming_check"]
|
||||||
_BLOCK_STAGES = ["fragment_filter", "dedup", "grouping", "gap_check", "done"]
|
_BLOCK_STAGES = ["fragment_filter", "dedup", "grouping", "gap_check", "done"]
|
||||||
_ART_STAGES = ["subblocks", "facts", "levels", "relevance", "question_pattern",
|
_ART_STAGES = ["subblocks", "facts", "levels", "relevance", "konsolidierung", "question_pattern",
|
||||||
"artefacts", "finalize", "outline"]
|
"artefacts", "finalize", "outline"]
|
||||||
# where a requeued dead card restarts, by kind
|
# where a requeued dead card restarts, by kind
|
||||||
_DEAD_RESTART = {"title": "cluster", "cluster": "pair_check", "block": "fragment_filter",
|
_DEAD_RESTART = {"title": "cluster", "cluster": "pair_check", "block": "fragment_filter",
|
||||||
@@ -1585,7 +1775,36 @@ async def board_snapshot(topic: str, limit: int = 20) -> dict:
|
|||||||
"title": r["payload"].get("title") or r["card_id"], "error": r.get("last_error") or ""}
|
"title": r["payload"].get("title") or r["card_id"], "error": r.get("last_error") or ""}
|
||||||
for r in await db.kanban_dead(topic)]
|
for r in await db.kanban_dead(topic)]
|
||||||
return {"columns": columns, "dead": dead,
|
return {"columns": columns, "dead": dead,
|
||||||
"done": counts.get("inventory", {}).get("done_block", 0)}
|
"done": counts.get("inventory", {}).get("done_block", 0),
|
||||||
|
"qa": _qa_view(topic, counts, flow)}
|
||||||
|
|
||||||
|
|
||||||
|
def _qa_view(topic: str, counts: dict, flow) -> dict | None:
|
||||||
|
"""Latest QA report digest for the board header. `pausiert` = the gate stopped the
|
||||||
|
flow (score below threshold, board-2 cards waiting, no flow running)."""
|
||||||
|
# no inventory (deleted/never built) → no badge; the report files stay on purpose,
|
||||||
|
# so the first run after a rebuild diffs against the old state
|
||||||
|
if not counts.get("inventory", {}).get("done_block", 0) and not (
|
||||||
|
flow and flow.state.get("qa_note") is not None):
|
||||||
|
return None
|
||||||
|
import qa
|
||||||
|
tdir = qa.QA_DIR / topic
|
||||||
|
# by mtime: a re-run overwrites the run-id-named file, which sorts before timestamp names.
|
||||||
|
# guide-* reports are the guide_qa series — they must not shadow the inventory badge.
|
||||||
|
reports = sorted((p for p in tdir.glob("*.json") if not p.name.startswith("guide-")),
|
||||||
|
key=lambda p: p.stat().st_mtime) if tdir.is_dir() else []
|
||||||
|
if not reports:
|
||||||
|
return None
|
||||||
|
r = _json_file(reports[-1]) or {}
|
||||||
|
note = r.get("note")
|
||||||
|
if note is None:
|
||||||
|
return None
|
||||||
|
wartend = counts.get("artefacts", {}).get("subblocks", 0)
|
||||||
|
pausiert = bool(note < QA_GATE_NOTE and wartend and flow is None)
|
||||||
|
return {"note": note, "note_artefakte": r.get("note_artefakte"),
|
||||||
|
"schwelle": QA_GATE_NOTE, "pausiert": pausiert,
|
||||||
|
"quoten": r.get("quoten", {}),
|
||||||
|
"befunde": (r.get("fremd", []) + r.get("unecht", []))[:6]}
|
||||||
|
|
||||||
|
|
||||||
async def _clean_artefact_state(topic: str, files: dict) -> None:
|
async def _clean_artefact_state(topic: str, files: dict) -> None:
|
||||||
|
|||||||
@@ -70,6 +70,11 @@ SUB_VARIANT_COS = 0.90
|
|||||||
# subs are statements: true covers measure 0.27–0.38 while a wrong hit measured 0.76. The
|
# subs are statements: true covers measure 0.27–0.38 while a wrong hit measured 0.76. The
|
||||||
# embedding stage only backs up the lexical one (catches „Line Breaks (Soft)" 0.888).
|
# embedding stage only backs up the lexical one (catches „Line Breaks (Soft)" 0.888).
|
||||||
SEED_COVER_COS = 0.80
|
SEED_COVER_COS = 0.80
|
||||||
|
# Sub duplicate CANDIDATE floor for the judge paths (in-block consolidation band hint,
|
||||||
|
# cross-block stage, QA detector): the bulk of real paraphrase duplicates measures 0.75–0.90
|
||||||
|
# (Markdown: 50 pairs in the band, 4 above) — below every auto-merge threshold, so an LLM
|
||||||
|
# judge decides. Candidates only; a merge still needs judge unanimity.
|
||||||
|
SUB_DUP_KANDIDAT_COS = 0.75
|
||||||
|
|
||||||
# Umbrella grouping (block granularity level 2, step "Blocks-Gruppierung", AFTER the filter):
|
# Umbrella grouping (block granularity level 2, step "Blocks-Gruppierung", AFTER the filter):
|
||||||
# collapse sibling DEFINITIONS that are components of ONE umbrella concept (TM model:
|
# collapse sibling DEFINITIONS that are components of ONE umbrella concept (TM model:
|
||||||
@@ -134,6 +139,11 @@ CRAWL_MIN_CHARS = 400 # too little te
|
|||||||
QUELLE_RELEVANZ_CHUNK = 12 # pages per rater package (small, since a snippet ships per page)
|
QUELLE_RELEVANZ_CHUNK = 12 # pages per rater package (small, since a snippet ships per page)
|
||||||
QUELLE_RELEVANZ_SNIPPET = 800 # body characters per page in the prompt (URL is the primary signal)
|
QUELLE_RELEVANZ_SNIPPET = 800 # body characters per page in the prompt (URL is the primary signal)
|
||||||
|
|
||||||
|
# QA gate: after the inventory phase an automatic QA run scores the blocks; below the
|
||||||
|
# threshold the flow PAUSES before board 2 burns tokens (frontend offers force-continue).
|
||||||
|
QA_GATE_NOTE = 9.5 # 0 = gate off; quota-based, so the tolerated finding count scales with topic size
|
||||||
|
QA_GATE_LLM = True # include the LLM samples (Echtheit/Dubletten) in the gate run
|
||||||
|
|
||||||
# Inline evidence for judge agents: corpus excerpts go INTO the prompt instead of letting
|
# Inline evidence for judge agents: corpus excerpts go INTO the prompt instead of letting
|
||||||
# every judge re-search the source folder (measured: ~10 tool turns/judge, 82 % of the
|
# every judge re-search the source folder (measured: ~10 tool turns/judge, 82 % of the
|
||||||
# run's tokens were cache reads from those loops).
|
# run's tokens were cache reads from those loops).
|
||||||
@@ -153,6 +163,8 @@ TIMEOUTS = {
|
|||||||
"content_check": (300, 10), # content exam per block in the package
|
"content_check": (300, 10), # content exam per block in the package
|
||||||
"subblock": (400, 15), # finder round — p95 measured 124 s (was 900+45n)
|
"subblock": (400, 15), # finder round — p95 measured 124 s (was 900+45n)
|
||||||
"subblock_check": (300, 15), # judge decides contested subblocks in the chunk
|
"subblock_check": (300, 15), # judge decides contested subblocks in the chunk
|
||||||
|
"konsolidierung": (600, 25), # consolidation judge sees ALL subs with key points — 585 s
|
||||||
|
# (subblock_check at n=19) produced 15 % timeouts
|
||||||
"level": (300, 10), # classify subblocks per chunk
|
"level": (300, 10), # classify subblocks per chunk
|
||||||
"level_check": (300, 10), # judge decides contested levels in the chunk
|
"level_check": (300, 10), # judge decides contested levels in the chunk
|
||||||
"relevance": (300, 10), # subblocks relevant/peripheral per chunk
|
"relevance": (300, 10), # subblocks relevant/peripheral per chunk
|
||||||
|
|||||||
@@ -213,7 +213,8 @@ CREATE TABLE IF NOT EXISTS events (
|
|||||||
status TEXT NOT NULL DEFAULT '',
|
status TEXT NOT NULL DEFAULT '',
|
||||||
dur_ms INTEGER,
|
dur_ms INTEGER,
|
||||||
wait_ms INTEGER,
|
wait_ms INTEGER,
|
||||||
meta TEXT NOT NULL DEFAULT '{}'
|
meta TEXT NOT NULL DEFAULT '{}',
|
||||||
|
run_id TEXT NOT NULL DEFAULT ''
|
||||||
)
|
)
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@@ -354,6 +355,10 @@ async def init_db():
|
|||||||
await db.execute("DROP TABLE IF EXISTS vertiefungen")
|
await db.execute("DROP TABLE IF EXISTS vertiefungen")
|
||||||
await db.execute("DROP TABLE IF EXISTS block_texte")
|
await db.execute("DROP TABLE IF EXISTS block_texte")
|
||||||
await db.execute("DROP TABLE IF EXISTS guide_progress")
|
await db.execute("DROP TABLE IF EXISTS guide_progress")
|
||||||
|
try: # migration: run_id per generation run (QA groups events by it)
|
||||||
|
await db.execute("ALTER TABLE events ADD COLUMN run_id TEXT NOT NULL DEFAULT ''")
|
||||||
|
except aiosqlite.OperationalError:
|
||||||
|
pass
|
||||||
await db.execute(
|
await db.execute(
|
||||||
"UPDATE guides SET status = 'error', progress = NULL, error_msg = 'Server restart' "
|
"UPDATE guides SET status = 'error', progress = NULL, error_msg = 'Server restart' "
|
||||||
"WHERE status IN ('queued', 'generating')"
|
"WHERE status IN ('queued', 'generating')"
|
||||||
@@ -722,14 +727,27 @@ async def kanban_advance(topic: str, board: str, card_id: str, stage: str) -> No
|
|||||||
await kanban_advance_many(topic, board, [(card_id, stage)])
|
await kanban_advance_many(topic, board, [(card_id, stage)])
|
||||||
|
|
||||||
|
|
||||||
|
# Current generation run per topic — every event writer stamps run_id from here, so no
|
||||||
|
# signature threading through agents/kanban is needed. Set/cleared by the flow entries
|
||||||
|
# (board_inventory.run_boards, guide_board.run_guide_board).
|
||||||
|
_current_run: dict[str, str] = {}
|
||||||
|
|
||||||
|
|
||||||
|
def set_current_run(topic: str, run_id: str | None) -> None:
|
||||||
|
if run_id:
|
||||||
|
_current_run[topic] = run_id
|
||||||
|
else:
|
||||||
|
_current_run.pop(topic, None)
|
||||||
|
|
||||||
|
|
||||||
async def add_event(topic: str, kind: str, key: str = "", label: str = "", status: str = "",
|
async def add_event(topic: str, kind: str, key: str = "", label: str = "", status: str = "",
|
||||||
dur_ms: int | None = None, wait_ms: int | None = None, meta: dict | None = None) -> None:
|
dur_ms: int | None = None, wait_ms: int | None = None, meta: dict | None = None) -> None:
|
||||||
"""One pipeline-history row, own commit. Callers treat this as fire-and-forget."""
|
"""One pipeline-history row, own commit. Callers treat this as fire-and-forget."""
|
||||||
db = await get_db()
|
db = await get_db()
|
||||||
await db.execute(
|
await db.execute(
|
||||||
"INSERT INTO events (topic, ts, kind, key, label, status, dur_ms, wait_ms, meta) VALUES (?,?,?,?,?,?,?,?,?)",
|
"INSERT INTO events (topic, ts, kind, key, label, status, dur_ms, wait_ms, meta, run_id) VALUES (?,?,?,?,?,?,?,?,?,?)",
|
||||||
(topic, _now(), kind, key, label, status, dur_ms, wait_ms,
|
(topic, _now(), kind, key, label, status, dur_ms, wait_ms,
|
||||||
json.dumps(meta or {}, ensure_ascii=False)))
|
json.dumps(meta or {}, ensure_ascii=False), _current_run.get(topic, "")))
|
||||||
await db.commit()
|
await db.commit()
|
||||||
|
|
||||||
|
|
||||||
@@ -737,9 +755,10 @@ async def _add_events_many(db, topic: str, rows: list[tuple]) -> None:
|
|||||||
"""Batch insert WITHOUT commit — must run inside the caller's transaction
|
"""Batch insert WITHOUT commit — must run inside the caller's transaction
|
||||||
(kanban_advance_many) so the event batch stays atomic with the moves."""
|
(kanban_advance_many) so the event batch stays atomic with the moves."""
|
||||||
now = _now()
|
now = _now()
|
||||||
|
rid = _current_run.get(topic, "")
|
||||||
await db.executemany(
|
await db.executemany(
|
||||||
"INSERT INTO events (topic, ts, kind, key, label, status, dur_ms, wait_ms, meta) VALUES (?,?,?,?,?,?,?,?,?)",
|
"INSERT INTO events (topic, ts, kind, key, label, status, dur_ms, wait_ms, meta, run_id) VALUES (?,?,?,?,?,?,?,?,?,?)",
|
||||||
[(topic, now, kind, key, label, status, None, None, "{}")
|
[(topic, now, kind, key, label, status, None, None, "{}", rid)
|
||||||
for kind, key, label, status in rows])
|
for kind, key, label, status in rows])
|
||||||
|
|
||||||
|
|
||||||
@@ -828,13 +847,39 @@ async def kanban_fail_card(topic: str, board: str, card_id: str, error: str,
|
|||||||
(retries, _now_plus(backoff_base * (2 ** (retries - 1))), error[:500], _now(),
|
(retries, _now_plus(backoff_base * (2 ** (retries - 1))), error[:500], _now(),
|
||||||
topic, board, card_id))
|
topic, board, card_id))
|
||||||
await db.execute(
|
await db.execute(
|
||||||
"INSERT INTO events (topic, ts, kind, key, label, status, dur_ms, wait_ms, meta) VALUES (?,?,?,?,?,?,?,?,?)",
|
"INSERT INTO events (topic, ts, kind, key, label, status, dur_ms, wait_ms, meta, run_id) VALUES (?,?,?,?,?,?,?,?,?,?)",
|
||||||
(topic, _now(), "fail", f"{board}:{card_id}", "", "dead" if dead else f"retry{retries}",
|
(topic, _now(), "fail", f"{board}:{card_id}", "", "dead" if dead else f"retry{retries}",
|
||||||
None, None, json.dumps({"error": error[:200]}, ensure_ascii=False)))
|
None, None, json.dumps({"error": error[:200]}, ensure_ascii=False), _current_run.get(topic, "")))
|
||||||
await db.commit()
|
await db.commit()
|
||||||
return dead
|
return dead
|
||||||
|
|
||||||
|
|
||||||
|
async def events_run_summary(topic: str, run_id: str) -> dict:
|
||||||
|
"""Agent/token aggregate of ONE run — the numbers block of lauf-summary.json."""
|
||||||
|
db = await get_db()
|
||||||
|
cursor = await db.execute(
|
||||||
|
"""SELECT status, COUNT(*), SUM(dur_ms),
|
||||||
|
SUM(json_extract(meta,'$.tokens.input')), SUM(json_extract(meta,'$.tokens.output')),
|
||||||
|
SUM(json_extract(meta,'$.tokens.cache_read')), SUM(json_extract(meta,'$.tokens.cache_write'))
|
||||||
|
FROM events WHERE topic = ? AND run_id = ? AND kind = 'agent' GROUP BY status""",
|
||||||
|
(topic, run_id))
|
||||||
|
agents = {"gesamt": 0, "ok": 0, "timeout": 0, "cancelled": 0, "sonstige": 0, "verlorene_min": 0}
|
||||||
|
tokens = {"input": 0, "output": 0, "cache_read": 0, "cache_write": 0}
|
||||||
|
for status, n, dur, ti, to, cr, cw in await cursor.fetchall():
|
||||||
|
agents["gesamt"] += n
|
||||||
|
if status in ("ok", "timeout", "cancelled"):
|
||||||
|
agents[status] += n
|
||||||
|
else:
|
||||||
|
agents["sonstige"] += n
|
||||||
|
if status != "ok":
|
||||||
|
agents["verlorene_min"] += round((dur or 0) / 60000)
|
||||||
|
tokens["input"] += ti or 0
|
||||||
|
tokens["output"] += to or 0
|
||||||
|
tokens["cache_read"] += cr or 0
|
||||||
|
tokens["cache_write"] += cw or 0
|
||||||
|
return {"agents": agents, "tokens": tokens}
|
||||||
|
|
||||||
|
|
||||||
async def kanban_dead(topic: str) -> list[dict]:
|
async def kanban_dead(topic: str) -> list[dict]:
|
||||||
"""Dead-letter cards across boards (for the board UI + requeue)."""
|
"""Dead-letter cards across boards (for the board UI + requeue)."""
|
||||||
return await kanban_cards(topic, stage="dead")
|
return await kanban_cards(topic, stage="dead")
|
||||||
@@ -873,6 +918,14 @@ async def kanban_stage_cards(topic: str, board: str, stage: str, limit: int = 20
|
|||||||
return [_card(row, cursor) for row in await cursor.fetchall()]
|
return [_card(row, cursor) for row in await cursor.fetchall()]
|
||||||
|
|
||||||
|
|
||||||
|
async def kanban_delete_card(topic: str, board: str, card_id: str) -> None:
|
||||||
|
"""Delete ONE card (repair: the merged-away/removed block's board-2 card)."""
|
||||||
|
db = await get_db()
|
||||||
|
await db.execute("DELETE FROM kanban_cards WHERE topic = ? AND board = ? AND card_id = ?",
|
||||||
|
(topic, board, card_id))
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
|
||||||
async def kanban_delete_cards(topic: str, board: str, kind: str | None = None) -> None:
|
async def kanban_delete_cards(topic: str, board: str, kind: str | None = None) -> None:
|
||||||
"""Delete derived cards (board reset) — kind=None wipes the whole board."""
|
"""Delete derived cards (board reset) — kind=None wipes the whole board."""
|
||||||
db = await get_db()
|
db = await get_db()
|
||||||
@@ -928,11 +981,16 @@ async def upsert_guide_card(topic: str, format: str, block_norm: str, block: str
|
|||||||
await db.commit()
|
await db.commit()
|
||||||
|
|
||||||
|
|
||||||
async def list_guide_cards(topic: str, format: str) -> list[dict]:
|
async def list_guide_cards(topic: str, format: str | None = None) -> list[dict]:
|
||||||
|
"""format=None: alle Formate — das Guide-QA misst den Bestand topic-weit."""
|
||||||
db = await get_db()
|
db = await get_db()
|
||||||
cursor = await db.execute(
|
if format is None:
|
||||||
"SELECT * FROM guide_cards WHERE topic = ? AND format = ? ORDER BY ord, block_norm",
|
cursor = await db.execute(
|
||||||
(topic, format))
|
"SELECT * FROM guide_cards WHERE topic = ? ORDER BY format, ord, block_norm", (topic,))
|
||||||
|
else:
|
||||||
|
cursor = await db.execute(
|
||||||
|
"SELECT * FROM guide_cards WHERE topic = ? AND format = ? ORDER BY ord, block_norm",
|
||||||
|
(topic, format))
|
||||||
return [_row_to_dict(row, cursor) for row in await cursor.fetchall()]
|
return [_row_to_dict(row, cursor) for row in await cursor.fetchall()]
|
||||||
|
|
||||||
|
|
||||||
@@ -1384,6 +1442,22 @@ async def subs_per_level_norm(topic: str) -> dict[str, dict[int, int]]:
|
|||||||
return out
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def delete_artefakt_row(topic: str, block_norm: str, sub_norm: str, type: str) -> None:
|
||||||
|
"""Remove ONE artefact row (repair: dead target — sub discarded or gone)."""
|
||||||
|
db = await get_db()
|
||||||
|
await db.execute("DELETE FROM sub_artefakte WHERE topic = ? AND block_norm = ? AND sub_norm = ? AND type = ?",
|
||||||
|
(topic, block_norm, sub_norm, type))
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
|
||||||
|
async def delete_frage_row(topic: str, block_norm: str, sub_norm: str) -> None:
|
||||||
|
"""Remove ONE question_pattern row (repair: dead target)."""
|
||||||
|
db = await get_db()
|
||||||
|
await db.execute("DELETE FROM question_pattern WHERE topic = ? AND block_norm = ? AND sub_norm = ?",
|
||||||
|
(topic, block_norm, sub_norm))
|
||||||
|
await db.commit()
|
||||||
|
|
||||||
|
|
||||||
async def delete_sub_artefakte(topic: str, block_norm: str | None = None) -> None:
|
async def delete_sub_artefakte(topic: str, block_norm: str | None = None) -> None:
|
||||||
db = await get_db()
|
db = await get_db()
|
||||||
if block_norm is None:
|
if block_norm is None:
|
||||||
|
|||||||
126
backend/fragen-vorrat-aak.md.cleaned.md
Normal file
126
backend/fragen-vorrat-aak.md.cleaned.md
Normal file
@@ -0,0 +1,126 @@
|
|||||||
|
# Frage-Muster für Lern-Prüfung: aak
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## BAUSTEIN: CliqueAndIndependentSet-Problem
|
||||||
|
|
||||||
|
### Subbaustein: Clique: Knotenmenge, in der je zwei Knoten durch eine Kante verbunden sind
|
||||||
|
**Muster:** Wann ist eine Knotenmenge C ⊆ V eine Clique in einem Graphen G = (V, E) und welche Bedingung müssen alle Knotenpaare einer Clique erfüllen?
|
||||||
|
|
||||||
|
### Subbaustein: Independent Set: Knotenmenge ohne Kanten zwischen je zwei Knoten
|
||||||
|
**Muster:** Was ist die formale Definition eines Independent Set in einem Graphen G = (V, E) und welche Bedingung muss für je zwei Knoten eines Independent Set gelten?
|
||||||
|
|
||||||
|
### Subbaustein: Komplementgraph: Independent Set in G ist Clique in G̅
|
||||||
|
**Muster:** Welche Beziehung besteht zwischen einem Independent Set in G und einer Clique in G̅, warum sind die Probleme gegenseitig in Polynomialzeit aufeinander reduzierbar, und was bleibt bei der Bildung des Komplementgraphen gleich bzw. ändert sich?
|
||||||
|
|
||||||
|
### Subbaustein: K-CLIQUE ⪯ K-INDEPENDENT-SET mittels Komplementgraph
|
||||||
|
**Muster:** Wie transformiert man eine Instanz (G, k) von CLIQUE in eine Instanz von INDEPENDENT-SET, und bleibt die Größe k bei der Reduktion erhalten?
|
||||||
|
|
||||||
|
### Subbaustein: NP-vollständig via gegenseitige Reduktion über Komplementgraph
|
||||||
|
**Muster:** Wie folgt aus Korollar 6.18 die NP-Vollständigkeit von Independent Set, und welche untere Schranke für die Laufzeit von Algorithmen für Independent Set folgt aus der ETH?
|
||||||
|
|
||||||
|
### Subbaustein: Existenz von IS bzw. CLIQUE der Größe k ist NP-vollständig
|
||||||
|
**Muster:** Durch welche Polynomialzeitreduktion lässt sich zeigen, dass Independent Set NP-schwer ist, und wie wird in Satz 6.26 die NP-Schwere von k-Clique bewiesen?
|
||||||
|
|
||||||
|
### Subbaustein: Beide Probleme sind in NP (Verifizierer existiert)
|
||||||
|
**Muster:** Welche Eigenschaft müssen Zertifikat und Verifizierer für die Probleme k-Clique und k-Independent-Set erfüllen?
|
||||||
|
|
||||||
|
### Subbaustein: Konsequenz: P = NP falls eines in P
|
||||||
|
**Muster:** Welche fundamentale Konsequenz ergibt sich aus Satz 6.16, wenn ein NP-vollständiges Problem in P liegt?
|
||||||
|
|
||||||
|
### Subbaustein: Formale Sprachen: CLIQUE und INDEPENDENT-SET
|
||||||
|
**Muster:** Wie sind die formalen Sprachen CLIQUE und INDEPENDENT-SET über dem Alphabet Σ = {0, 1} kodiert und welche Struktur haben sie?
|
||||||
|
|
||||||
|
### Subbaustein: Eingabe/Ausgabe von k-Clique und k-Independent-Set
|
||||||
|
**Muster:** Was ist die Eingabe und was die Ausgabe bei den Entscheidungsproblemen k-Clique und k-Independent-Set?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## BAUSTEIN: Reduktion CLIQUE → CLIQUE-NOMEMBER
|
||||||
|
|
||||||
|
### Subbaustein: Füge isolierten Knoten v zu G hinzu: G' = G ∪ {v}
|
||||||
|
**Muster:** Wie wird bei der Reduktion von CLIQUE auf CLIQUE-NOMEMBER der neue Graph G' konstruiert, und welche Elemente werden gegenüber der ursprünglichen Instanz verändert?
|
||||||
|
|
||||||
|
### Subbaustein: v ist in G' in keiner k-Clique (isoliert)
|
||||||
|
**Muster:** Warum kann der hinzugefügte Knoten v in keiner gültigen k-Clique von G' enthalten sein, und welche Eigenschaft hat der Knoten v in der konstruierten Instanz (G', v, k)?
|
||||||
|
|
||||||
|
### Subbaustein: G hat k-Clique ⟺ G' hat (k+1)-Clique mit v
|
||||||
|
**Muster:** Wie hängt eine k-Clique in G mit einer k-Clique in G' zusammen, und warum bleibt die Cliquengröße k bei der Reduktion unverändert?
|
||||||
|
|
||||||
|
### Subbaustein: Polynomielle Transformation
|
||||||
|
**Muster:** Warum ist die beschriebene Reduktion von CLIQUE auf CLIQUE-NOMEMBER in polynomieller Zeit berechenbar?
|
||||||
|
|
||||||
|
### Subbaustein: CLIQUE: Eingabe Graph G, Frage: existiert K-clique?
|
||||||
|
**Muster:** Was ist die Eingabe und was ist die Frage beim Entscheidungsproblem CLIQUE?
|
||||||
|
|
||||||
|
### Subbaustein: CLIQUE-NOMEMBER formal definiert
|
||||||
|
**Muster:** Wie ist das Problem CLIQUE-NOMEMBER gemäß Skript 6.50 formal definiert?
|
||||||
|
|
||||||
|
### Subbaustein: Reduktion beweist CLIQUE-NOMEMBER ∈ NP-vollständig
|
||||||
|
**Muster:** Welche drei Bedingungen müssen erfüllt sein, damit CLIQUE-NOMEMBER als NP-vollständig gilt?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## BAUSTEIN: Independent Set
|
||||||
|
|
||||||
|
### Subbaustein: Independent Set S⊆V: keine Kante zwischen je zwei Knoten in S
|
||||||
|
**Muster:** Welche Bedingung muss für je zwei Knoten eines Independent Set gelten und was bedeutet es, dass die Knoten eines Independent Set paarweise nicht adjazent sind?
|
||||||
|
|
||||||
|
### Subbaustein: Komplementär zur Clique
|
||||||
|
**Muster:** In welchem Graphen entspricht ein Independent Set einer Clique und wie hängt ein Independent Set in G mit einer Clique im Komplementgraphen G' zusammen?
|
||||||
|
|
||||||
|
### Subbaustein: NP-vollständiges Problem
|
||||||
|
**Muster:** Welche Komplexitätsklasse enthält Independent Set und wie wurde dies bewiesen?
|
||||||
|
|
||||||
|
### Subbaustein: INDEPENDENT-SET = {(G,k) | G enthält unabhängige Menge der Größe ≥k}
|
||||||
|
**Muster:** Welche Sprache formalisiert das Entscheidungsproblem Independent Set?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## BAUSTEIN: Tiefensuche (DFS) für Zykluserkennung
|
||||||
|
|
||||||
|
### Subbaustein: Weiß/Grau/Schwarz: Farbcodierung der DFS
|
||||||
|
**Muster:** Welche Farbe hat ein Knoten während er von der DFS bearbeitet wird, welche nach Abschluss, und wann wird ein Knoten in der DFS schwarz gefärbt?
|
||||||
|
|
||||||
|
### Subbaustein: Tree Edge (weiß): Kante zu unbesuchtem Knoten
|
||||||
|
**Muster:** Welche Kante wird als Tree Edge bezeichnet?
|
||||||
|
|
||||||
|
### Subbaustein: Rückkante (grau → weiß): signalisiert Zyklus
|
||||||
|
**Muster:** Zu einem Knoten welcher Farbe muss eine Kante führen, um einen Zyklus anzuzeigen?
|
||||||
|
|
||||||
|
### Subbaustein: DFS-Zykluserkennung in O(V+E) bei adjacency List
|
||||||
|
**Muster:** Warum beträgt die Laufzeit der DFS-Zykluserkennung bei Adjazenzliste Θ(|V|+|E|)?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## BAUSTEIN: Turingmaschine für 0^n (Zweierpotenz)
|
||||||
|
|
||||||
|
### Subbaustein: Eingabe: n Nullen in unärer Codierung
|
||||||
|
**Muster:** In welcher Codierung wird die Eingabezahl n der TM für 0^n dargestellt?
|
||||||
|
|
||||||
|
### Subbaustein: Akzeptiert nur wenn n = 2^k für ein k ≥ 0
|
||||||
|
**Muster:** Nach welchem Kriterium entscheidet die TM, ob eine Eingabe akzeptiert wird?
|
||||||
|
|
||||||
|
### Subbaustein: Phase 1: Markiere jede zweite 0 mit x (alternierend)
|
||||||
|
**Muster:** Wie markiert die TM die Nullen im ersten Schritt?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## BAUSTEIN: 3-SAT zu 3-Färbung Reduktion
|
||||||
|
|
||||||
|
### Subbaustein: Knotenzahl linear in Variablen und Klauseln
|
||||||
|
**Muster:** Aus welchen Komponenten setzt sich die Knotenmenge V der konstruierten Instanz zusammen?
|
||||||
|
|
||||||
|
### Subbaustein: Dreieck erzwingt drei verschiedene Farben für die drei Knoten
|
||||||
|
**Muster:** Warum benötigen die drei Knoten xi, x̄i und vi eines jeden Dreiecks drei verschiedene Farben?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## BAUSTEIN: MC-Knapsack
|
||||||
|
|
||||||
|
### Subbaustein: Ziel: Maximierung des Gesamtwerts
|
||||||
|
**Muster:** Was ist die Zielfunktion beim Maximum-Cut Knapsack Problem?
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Gesamt: 28 Frage-Muster**
|
||||||
@@ -20,6 +20,7 @@ import re
|
|||||||
|
|
||||||
import database as db
|
import database as db
|
||||||
import readability
|
import readability
|
||||||
|
from blocks import _sink_json
|
||||||
from config import FORMAT_PURPOSE, READABILITY_ACTIVE, TEMPLATES_DIR, MAX_CONCURRENT_AGENTS_PER_TOPIC
|
from config import FORMAT_PURPOSE, READABILITY_ACTIVE, TEMPLATES_DIR, MAX_CONCURRENT_AGENTS_PER_TOPIC
|
||||||
from fsutil import atomic_write_json
|
from fsutil import atomic_write_json
|
||||||
from jsonio import read_json_file as _json_file
|
from jsonio import read_json_file as _json_file
|
||||||
@@ -195,22 +196,34 @@ async def _set(env: _Env, card: dict, **fields):
|
|||||||
async def _stage_lernziele(env: _Env, card: dict) -> bool:
|
async def _stage_lernziele(env: _Env, card: dict) -> bool:
|
||||||
norm = card["block_norm"]
|
norm = card["block_norm"]
|
||||||
if not await db.list_lernziele(env.topic, norm):
|
if not await db.list_lernziele(env.topic, norm):
|
||||||
path = env.slot(f"ziele-{_safe(norm)}.json")
|
|
||||||
subs = "\n".join(f"- [{s.get('level', 'beginner')}] {s['title']}"
|
subs = "\n".join(f"- [{s.get('level', 'beginner')}] {s['title']}"
|
||||||
for s in env.subs_by_title.get(card["block"], [])) or "(keine)"
|
for s in env.subs_by_title.get(card["block"], [])) or "(keine)"
|
||||||
status, ziele = await run_single_slot(
|
|
||||||
env.ctx, f"Lernziele {card['block']}", key=f"{env.guide_id}-ziele-{_safe(norm)}",
|
async def _versuch(suffix: str):
|
||||||
prompt=_prompt("Guide-Lernziele", topic=env.topic, block=card["block"],
|
path = env.slot(f"ziele-{_safe(norm)}{suffix}.json")
|
||||||
subs=subs, facts=_card_facts(env, card["block"]),
|
return await run_single_slot(
|
||||||
out_path=path, extra=_extra(env.instructions)),
|
env.ctx, f"Lernziele {card['block']}", key=f"{env.guide_id}-ziele-{_safe(norm)}{suffix}",
|
||||||
role="judge", capabilities="files",
|
prompt=_prompt("Guide-Lernziele", topic=env.topic, block=card["block"],
|
||||||
payload=lambda result: _ziele_schema(_json_file(path)),
|
subs=subs, facts=_card_facts(env, card["block"]),
|
||||||
timeout=_timeout("lernziele", len(env.subs_by_title.get(card["block"], []))))
|
out_path=path, extra=_extra(env.instructions)),
|
||||||
|
role="judge", capabilities="files",
|
||||||
|
payload=lambda result, p=path: _ziele_schema(_json_file(p)),
|
||||||
|
timeout=_timeout("lernziele", len(env.subs_by_title.get(card["block"], []))))
|
||||||
|
|
||||||
|
status, ziele = await _versuch("")
|
||||||
if status == CANCELLED:
|
if status == CANCELLED:
|
||||||
return False
|
return False
|
||||||
if status == FAILED:
|
if status == FAILED:
|
||||||
await _set(env, card, status="error", gate_info="Lernziele ohne Ergebnis")
|
await _set(env, card, status="error", gate_info="Lernziele ohne Ergebnis")
|
||||||
return False
|
return False
|
||||||
|
if not ziele: # ein Ersatz-Versuch — leere Liste heißt: das Coverage-Gate läuft leer
|
||||||
|
status, ziele = await _versuch("-2")
|
||||||
|
if status == CANCELLED:
|
||||||
|
return False
|
||||||
|
if not isinstance(ziele, list):
|
||||||
|
ziele = []
|
||||||
|
if not ziele:
|
||||||
|
_log(env.topic, f"Lernziele {card['block']}: zweimal leer — Block ohne Coverage-Gate")
|
||||||
for z in ziele:
|
for z in ziele:
|
||||||
await db.put_lernziel(env.topic, norm, z["id"], z["text"], _norm_title(z["sub"]))
|
await db.put_lernziel(env.topic, norm, z["id"], z["text"], _norm_title(z["sub"]))
|
||||||
await _set(env, card, stage="zuweisung", status="open")
|
await _set(env, card, stage="zuweisung", status="open")
|
||||||
@@ -254,6 +267,12 @@ def _merge_split_sections(sec_a: dict, sec_b: dict) -> str:
|
|||||||
return "\n\n".join(lines)
|
return "\n\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
def _writer_budget(n_subs: int, sockel: int = 800) -> int:
|
||||||
|
"""Length guideline (chars) for the detailed version — unguided sections measured 2–4×
|
||||||
|
too long (23k) or, after the readability fix, far too thin (180 chars/sub)."""
|
||||||
|
return sockel + 400 * max(n_subs, 1)
|
||||||
|
|
||||||
|
|
||||||
async def _write_split(env: _Env, card: dict, ziele_text: str):
|
async def _write_split(env: _Env, card: dict, ziele_text: str):
|
||||||
"""First draft in two halves (parallel), merged into one section.
|
"""First draft in two halves (parallel), merged into one section.
|
||||||
→ merged text | None (failed) | False (cancelled)."""
|
→ merged text | None (failed) | False (cancelled)."""
|
||||||
@@ -290,6 +309,7 @@ async def _write_split(env: _Env, card: dict, ziele_text: str):
|
|||||||
examples=await _card_examples(env, norm, parts[i],
|
examples=await _card_examples(env, norm, parts[i],
|
||||||
include_unmatched=(i == 0)),
|
include_unmatched=(i == 0)),
|
||||||
gaps="\n" + hints[i] + "\n",
|
gaps="\n" + hints[i] + "\n",
|
||||||
|
budget=_writer_budget(len(parts[i]), sockel=400),
|
||||||
spec=env.spec, out_path=path, extra=_extra(env.instructions)),
|
spec=env.spec, out_path=path, extra=_extra(env.instructions)),
|
||||||
role="guide", capabilities="files", payload=_payload,
|
role="guide", capabilities="files", payload=_payload,
|
||||||
timeout=_timeout("writer", 1))
|
timeout=_timeout("writer", 1))
|
||||||
@@ -338,6 +358,7 @@ async def _stage_writer(env: _Env, card: dict) -> bool:
|
|||||||
facts=_card_facts(env, card["block"]),
|
facts=_card_facts(env, card["block"]),
|
||||||
examples=await _card_examples(env, norm, env.subs_by_title.get(card["block"], [])),
|
examples=await _card_examples(env, norm, env.subs_by_title.get(card["block"], [])),
|
||||||
gaps=gaps, spec=env.spec,
|
gaps=gaps, spec=env.spec,
|
||||||
|
budget=_writer_budget(len(env.subs_by_title.get(card["block"], []))),
|
||||||
out_path=path, extra=_extra(env.instructions)),
|
out_path=path, extra=_extra(env.instructions)),
|
||||||
role="guide", capabilities="files", payload=_payload,
|
role="guide", capabilities="files", payload=_payload,
|
||||||
timeout=_timeout("writer", 1))
|
timeout=_timeout("writer", 1))
|
||||||
@@ -455,13 +476,15 @@ async def _stage_lesbarkeit(env: _Env, card: dict) -> bool:
|
|||||||
return False
|
return False
|
||||||
problems: list[str] = []
|
problems: list[str] = []
|
||||||
path = env.slot(f"lese-{_safe(norm)}-r{card['writer_rounds']}.json")
|
path = env.slot(f"lese-{_safe(norm)}-r{card['writer_rounds']}.json")
|
||||||
|
# Text-Antwort + Engine-Sink: Datei-schreibende Judges lieferten invalides JSON
|
||||||
|
# (3 kaputte Check-Dateien im Messlauf) — der Sink validiert vor dem Persistieren
|
||||||
status, res = await run_single_slot(
|
status, res = await run_single_slot(
|
||||||
env.ctx, f"Lese-Check {card['block']}", key=f"{env.guide_id}-lese-{_safe(norm)}",
|
env.ctx, f"Lese-Check {card['block']}", key=f"{env.guide_id}-lese-{_safe(norm)}",
|
||||||
prompt=_prompt("Guide-Lese-Check", topic=env.topic, format_name=env.format,
|
prompt=_prompt("Guide-Lese-Check", topic=env.topic, format_name=env.format,
|
||||||
spec=env.spec, sections=f"SECTION: {card['block']}\n{sec['md']}",
|
spec=env.spec, sections=f"SECTION: {card['block']}\n{sec['md']}",
|
||||||
out_path=path, extra=_extra(env.instructions)),
|
extra=_extra(env.instructions)),
|
||||||
role="judge", capabilities="files",
|
role="judge", capabilities="none",
|
||||||
payload=lambda result: _problems_schema(_json_file(path)),
|
payload=lambda result: _sink_json(result, path, _problems_schema),
|
||||||
timeout=_timeout("lese_check", 1))
|
timeout=_timeout("lese_check", 1))
|
||||||
if status == CANCELLED:
|
if status == CANCELLED:
|
||||||
return False
|
return False
|
||||||
@@ -477,6 +500,8 @@ async def _stage_lesbarkeit(env: _Env, card: dict) -> bool:
|
|||||||
sub_list = "\n".join(f"- [{_level_label(s)}] {s['title']}" for s in subs) or "(none)"
|
sub_list = "\n".join(f"- [{_level_label(s)}] {s['title']}" for s in subs) or "(none)"
|
||||||
tasks = (f"SECTION: {card['block']}\n"
|
tasks = (f"SECTION: {card['block']}\n"
|
||||||
f"SUBBLOCKS (set one `<!-- sub: LABEL | title -->` marker each, label/order as here):\n{sub_list}\n"
|
f"SUBBLOCKS (set one `<!-- sub: LABEL | title -->` marker each, label/order as here):\n{sub_list}\n"
|
||||||
|
f"LENGTH TARGET: about {_writer_budget(len(subs))} characters for the detailed "
|
||||||
|
f"version (guideline — covering every subblock beats brevity).\n"
|
||||||
f"PROBLEM: {' · '.join(problems)}\nCURRENT CONTENT:\n{sec['md']}")
|
f"PROBLEM: {' · '.join(problems)}\nCURRENT CONTENT:\n{sec['md']}")
|
||||||
fixp = env.slot(f"lesefix-{_safe(norm)}.md")
|
fixp = env.slot(f"lesefix-{_safe(norm)}.md")
|
||||||
fixp.unlink(missing_ok=True)
|
fixp.unlink(missing_ok=True)
|
||||||
@@ -562,6 +587,9 @@ async def run_guide_board(guide_id: str, topic: str, format_name: str, entries:
|
|||||||
from guide import _load_subblocks
|
from guide import _load_subblocks
|
||||||
ctx = GenContext(topic=topic, provider=provider,
|
ctx = GenContext(topic=topic, provider=provider,
|
||||||
is_cancelled=lambda: is_guide_cancelled(guide_id), guide_id=guide_id)
|
is_cancelled=lambda: is_guide_cancelled(guide_id), guide_id=guide_id)
|
||||||
|
import uuid
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
db.set_current_run(topic, f"{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M')}-g{uuid.uuid4().hex[:4]}")
|
||||||
spec = (TEMPLATES_DIR / "Format" / "Section.md").read_text(encoding="utf-8")
|
spec = (TEMPLATES_DIR / "Format" / "Section.md").read_text(encoding="utf-8")
|
||||||
subs_raw = await _load_subblocks(topic)
|
subs_raw = await _load_subblocks(topic)
|
||||||
project = source_folder(topic)
|
project = source_folder(topic)
|
||||||
@@ -590,6 +618,9 @@ async def run_guide_board(guide_id: str, topic: str, format_name: str, entries:
|
|||||||
await asyncio.gather(*[_run_card(env, c, sem) for c in open_cards])
|
await asyncio.gather(*[_run_card(env, c, sem) for c in open_cards])
|
||||||
finally:
|
finally:
|
||||||
reporter.cancel()
|
reporter.cancel()
|
||||||
|
db.set_current_run(topic, None)
|
||||||
|
else:
|
||||||
|
db.set_current_run(topic, None)
|
||||||
if is_guide_cancelled(guide_id):
|
if is_guide_cancelled(guide_id):
|
||||||
return None
|
return None
|
||||||
# assembly — identical shape to the legacy pipeline
|
# assembly — identical shape to the legacy pipeline
|
||||||
@@ -617,6 +648,14 @@ async def run_guide_board(guide_id: str, topic: str, format_name: str, entries:
|
|||||||
})
|
})
|
||||||
for ch in order:
|
for ch in order:
|
||||||
chapters.append({"title": ch, "sections": by_chapter[ch]})
|
chapters.append({"title": ch, "sections": by_chapter[ch]})
|
||||||
|
if chapters:
|
||||||
|
try: # Abschluss-Guide-QA (best-effort): speist das Badge mit einer frischen Note
|
||||||
|
import guide_qa
|
||||||
|
rep = await guide_qa.guide_qa_report(topic, llm=True)
|
||||||
|
if rep:
|
||||||
|
await asyncio.to_thread(guide_qa._write_report, rep)
|
||||||
|
except Exception:
|
||||||
|
log.exception("[%s] Abschluss-Guide-QA fehlgeschlagen", topic)
|
||||||
return chapters or None
|
return chapters or None
|
||||||
|
|
||||||
|
|
||||||
@@ -655,7 +694,11 @@ async def board_snapshot(topic: str, format_name: str, limit: int = 20) -> dict:
|
|||||||
"ziele": f"{zc[0]}/{zc[1]}" if zc else ""})
|
"ziele": f"{zc[0]}/{zc[1]}" if zc else ""})
|
||||||
columns.append({"key": stage, "label": STAGE_LABELS[stage],
|
columns.append({"key": stage, "label": STAGE_LABELS[stage],
|
||||||
"total": len(in_stage), "cards": views})
|
"total": len(in_stage), "cards": views})
|
||||||
return {"columns": columns}
|
import qa as qa_mod # lazy wie in board_inventory
|
||||||
|
tdir = qa_mod.QA_DIR / topic
|
||||||
|
greports = sorted(tdir.glob("guide-*.json"), key=lambda p: p.stat().st_mtime) if tdir.is_dir() else []
|
||||||
|
note_guide = (_json_file(greports[-1]) or {}).get("note_guide") if greports else None
|
||||||
|
return {"columns": columns, "qa_guide": note_guide}
|
||||||
|
|
||||||
|
|
||||||
async def reset_card(topic: str, format_name: str, block_norm: str, ab_stage: int) -> bool:
|
async def reset_card(topic: str, format_name: str, block_norm: str, ab_stage: int) -> bool:
|
||||||
|
|||||||
200
backend/guide_qa.py
Normal file
200
backend/guide_qa.py
Normal file
@@ -0,0 +1,200 @@
|
|||||||
|
"""Unabhängiges Guide-Audit über einen FERTIGEN Guide — read-only.
|
||||||
|
|
||||||
|
Misst den gebauten Guide (guide_cards) gegen Lernziele und Sub-Satz mit Detektoren,
|
||||||
|
die bewusst NICHT die Pipeline-Gates wiederverwenden (covered-Flag, Fakten-Gate) —
|
||||||
|
geteilte blinde Flecken machen das Audit wertlos. Geteilt nur Infra: DB, readability,
|
||||||
|
Agent-Runner (--llm), Note-Formel aus qa.py.
|
||||||
|
|
||||||
|
CLI: python3 guide_qa.py <topic> [--llm] (oder: make qa-guide TOPIC=<topic> [LLM=1])
|
||||||
|
Report: storage/qa/<topic>/guide-<ts>.json + Konsolen-Digest.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
import database as db
|
||||||
|
import qa
|
||||||
|
import readability
|
||||||
|
from fsutil import atomic_write_json
|
||||||
|
from textkit import _norm_title
|
||||||
|
|
||||||
|
LAENGE_MIN = 150 # Zeichen je relevantem Sub im ausführlich-Teil (Untergrenze)
|
||||||
|
LAENGE_MAX = 1200 # Obergrenze — außerhalb = Tiefen-Lotterie statt Zerlegungs-Signal
|
||||||
|
JACCARD_ABSATZ = 0.6 # Wort-Jaccard, ab dem zwei Absätze als Doppel gelten
|
||||||
|
ABSATZ_MIN_CHARS = 200 # kürzere Absätze sind Übergänge — kein Dubletten-Signal
|
||||||
|
LLM_SECTION_CHARS = 2500 # Section-Auszug je Judge-Item
|
||||||
|
# fachliche Fehler wiegen am schwersten; Anker-lose Ziele = Coverage-Behauptung ohne Text.
|
||||||
|
NOTE_GEWICHTE_GUIDE = {"fachlich_falsch": 3.0, "ziel_ohne_anker": 2.0, "marker_fehlend": 1.5,
|
||||||
|
"redundanz": 1.0, "laengen_ausreisser": 0.5, "lesbarkeit": 0.5}
|
||||||
|
|
||||||
|
_MARKER = re.compile(r"<!--\s*sub:\s*\w+\s*\|\s*(.*?)\s*-->")
|
||||||
|
|
||||||
|
|
||||||
|
def _ausfuehrlich(md: str) -> str:
|
||||||
|
"""Der Lern-Fließtext einer Karte (hinter dem ausführlich-Marker, sonst alles)."""
|
||||||
|
teile = re.split(r"<!--\s*ausführlich\s*-->", md or "", maxsplit=1)
|
||||||
|
return teile[1] if len(teile) == 2 else (md or "")
|
||||||
|
|
||||||
|
|
||||||
|
def marker_fehlend(cards: list[dict], subs_rel: dict[str, set]) -> list[str]:
|
||||||
|
"""Relevante Subs ohne Sub-Marker in der Section — der Level-Filter verliert sie."""
|
||||||
|
out = []
|
||||||
|
for c in cards:
|
||||||
|
marker = {_norm_title(m) for m in _MARKER.findall(c["md"] or "")}
|
||||||
|
for sn in sorted(subs_rel.get(c["block_norm"], set())):
|
||||||
|
if sn not in marker and not any(m.startswith(sn) or sn.startswith(m) for m in marker):
|
||||||
|
out.append(f"{c['block']} · {sn}")
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def ziel_ohne_anker(cards: list[dict], ziele: list[dict]) -> list[str]:
|
||||||
|
"""Lernziele, deren distinktive Tokens im Section-Text fehlen — eigener Anker-Check,
|
||||||
|
NICHT das covered-Flag der Pipeline (das hat der Coverage-Judge selbst gesetzt)."""
|
||||||
|
text_by_norm = {c["block_norm"]: qa._tokens(_ausfuehrlich(c["md"])) for c in cards}
|
||||||
|
out = []
|
||||||
|
for z in ziele:
|
||||||
|
toks = qa._distinctive(z["text"])
|
||||||
|
st = text_by_norm.get(z["block_norm"])
|
||||||
|
if st is None or not toks:
|
||||||
|
continue
|
||||||
|
if len(toks & st) < min(2, len(toks)):
|
||||||
|
out.append(f"{z['block_norm']} · ({z['ziel_id']}) {z['text'][:60]}")
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def laengen_ausreisser(cards: list[dict], subs_rel: dict[str, set]) -> list[dict]:
|
||||||
|
out = []
|
||||||
|
for c in cards:
|
||||||
|
n = max(len(subs_rel.get(c["block_norm"], set())), 1)
|
||||||
|
pro_sub = len(_ausfuehrlich(c["md"])) / n
|
||||||
|
if not (LAENGE_MIN <= pro_sub <= LAENGE_MAX):
|
||||||
|
out.append({"block": c["block"], "zeichen_pro_sub": round(pro_sub)})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def redundanz(cards: list[dict]) -> list[dict]:
|
||||||
|
"""Absatz-Paare topic-weit mit hoher Token-Überlappung — derselbe Stoff doppelt erklärt."""
|
||||||
|
absaetze = []
|
||||||
|
for c in cards:
|
||||||
|
for a in _ausfuehrlich(c["md"]).split("\n\n"):
|
||||||
|
a = a.strip()
|
||||||
|
if len(a) >= ABSATZ_MIN_CHARS:
|
||||||
|
absaetze.append((c["block"], a, qa._tokens(a)))
|
||||||
|
out = []
|
||||||
|
for i in range(len(absaetze)):
|
||||||
|
for j in range(i + 1, len(absaetze)):
|
||||||
|
if qa._jaccard(absaetze[i][2], absaetze[j][2]) >= JACCARD_ABSATZ:
|
||||||
|
out.append({"a": f"{absaetze[i][0]}: {absaetze[i][1][:60]}",
|
||||||
|
"b": f"{absaetze[j][0]}: {absaetze[j][1][:60]}"})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def lesbarkeit(cards: list[dict]) -> list[str]:
|
||||||
|
"""Deterministisches externes Rating; Modell nicht ladbar → nicht gemessen (zählt nicht)."""
|
||||||
|
try:
|
||||||
|
hints = readability.rate_sections({i: _ausfuehrlich(c["md"]) for i, c in enumerate(cards, 1)})
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
return [f"{cards[i - 1]['block']}: {h}" for i, h in sorted(hints.items()) if h]
|
||||||
|
|
||||||
|
|
||||||
|
async def _fachlich_falsch(topic: str, cards: list[dict]) -> list[str]:
|
||||||
|
"""LLM-Stichprobe: Section enthält eine fachlich falsche Aussage? (eigenes Template)."""
|
||||||
|
from agents import run_agent
|
||||||
|
from jsonio import parse_json_text
|
||||||
|
from pipeline import _yesno_schema
|
||||||
|
out = []
|
||||||
|
for lo in range(0, len(cards), 5):
|
||||||
|
chunk = cards[lo:lo + 5]
|
||||||
|
listing = "\n\n".join(f"{k}. SECTION {c['block']}:\n{_ausfuehrlich(c['md'])[:LLM_SECTION_CHARS]}"
|
||||||
|
for k, c in enumerate(chunk, 1))
|
||||||
|
rc, txt, _err = await run_agent(
|
||||||
|
f"qa-guide-{topic}-fakten-{lo}", qa._qa_prompt("QA-Guide-Fakten", topic=topic, extra="", sections=listing),
|
||||||
|
600, role="judge", capabilities="none", scope=topic, label=f"Guide-QA Fakten {lo}")
|
||||||
|
v = (_yesno_schema(parse_json_text(txt)) or {}) if rc == 0 else {}
|
||||||
|
out += [c["block"] for k, c in enumerate(chunk, 1) if v.get(k) == "ja"]
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def guide_qa_report(topic: str, llm: bool = False) -> dict | None:
|
||||||
|
cards = [dict(r) for r in await db.list_guide_cards(topic)]
|
||||||
|
cards = [c for c in cards if (c.get("md") or "").strip()]
|
||||||
|
if not cards:
|
||||||
|
print(f"Keine Guide-Karten für '{topic}' — Guide noch nicht gebaut?")
|
||||||
|
return None
|
||||||
|
subs_rel: dict[str, set] = {}
|
||||||
|
for r in await db.list_subblocks(topic):
|
||||||
|
if r["status"] == "consensus" and r["relevance"] != "peripheral":
|
||||||
|
subs_rel.setdefault(r["block_norm"], set()).add(r["sub_norm"])
|
||||||
|
ziele = [dict(r) for r in await db.list_lernziele(topic)]
|
||||||
|
|
||||||
|
mf = marker_fehlend(cards, subs_rel)
|
||||||
|
za = ziel_ohne_anker(cards, ziele)
|
||||||
|
la = laengen_ausreisser(cards, subs_rel)
|
||||||
|
rd = redundanz(cards)
|
||||||
|
lb = lesbarkeit(cards)
|
||||||
|
falsch = await _fachlich_falsch(topic, cards) if llm else None
|
||||||
|
|
||||||
|
n_subs = max(sum(len(s) for s in subs_rel.values()), 1)
|
||||||
|
n_abs = max(sum(len([a for a in _ausfuehrlich(c["md"]).split("\n\n") if len(a.strip()) >= ABSATZ_MIN_CHARS])
|
||||||
|
for c in cards), 1)
|
||||||
|
quoten = {
|
||||||
|
"marker_fehlend": round(len(mf) / n_subs, 3),
|
||||||
|
"ziel_ohne_anker": round(len(za) / max(len(ziele), 1), 3),
|
||||||
|
"laengen_ausreisser": round(len(la) / len(cards), 3),
|
||||||
|
"redundanz": round(len(rd) / n_abs, 3),
|
||||||
|
"lesbarkeit": round(len(lb) / len(cards), 3),
|
||||||
|
**({"fachlich_falsch": round(len(falsch) / len(cards), 3)} if falsch is not None else {}),
|
||||||
|
}
|
||||||
|
report = {
|
||||||
|
"topic": topic, "erstellt": datetime.now(timezone.utc).isoformat(), "art": "guide",
|
||||||
|
"bloecke": len(cards), "ziele": len(ziele),
|
||||||
|
"quoten": quoten, "note_guide": qa.note(quoten, NOTE_GEWICHTE_GUIDE),
|
||||||
|
"marker_fehlend": mf, "ziel_ohne_anker": za, "laengen_ausreisser": la,
|
||||||
|
"redundanz": rd[:20], "lesbarkeit": lb,
|
||||||
|
**({"fachlich_falsch": falsch} if falsch is not None else {}),
|
||||||
|
"note_gewichte": NOTE_GEWICHTE_GUIDE,
|
||||||
|
}
|
||||||
|
return report
|
||||||
|
|
||||||
|
|
||||||
|
def _write_report(report: dict):
|
||||||
|
tdir = qa.QA_DIR / report["topic"]
|
||||||
|
tdir.mkdir(parents=True, exist_ok=True)
|
||||||
|
path = tdir / f"guide-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}.json"
|
||||||
|
atomic_write_json(path, report, indent=1)
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def _digest(report: dict, path):
|
||||||
|
print(f"Guide-QA {report['topic']} — {report['bloecke']} Sections, {report['ziele']} Ziele"
|
||||||
|
f" — Note {report['note_guide']}/10")
|
||||||
|
for k, v in report["quoten"].items():
|
||||||
|
print(f" {k:20} {v:6.1%}")
|
||||||
|
for k in ("marker_fehlend", "ziel_ohne_anker", "lesbarkeit", "fachlich_falsch"):
|
||||||
|
for x in report.get(k, [])[:5]:
|
||||||
|
print(f" {k.upper():16} {str(x)[:90]}")
|
||||||
|
for p in report.get("redundanz", [])[:5]:
|
||||||
|
print(f" DOPPELT? {p['a'][:55]} <-> {p['b'][:55]}")
|
||||||
|
print(f"Report: {path}")
|
||||||
|
|
||||||
|
|
||||||
|
async def main(topic: str, llm: bool):
|
||||||
|
await db.init_db()
|
||||||
|
try:
|
||||||
|
report = await guide_qa_report(topic, llm=llm)
|
||||||
|
if report is None:
|
||||||
|
sys.exit(1)
|
||||||
|
_digest(report, _write_report(report))
|
||||||
|
finally:
|
||||||
|
await db.close_db()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||||
|
if not args:
|
||||||
|
print("Nutzung: python3 guide_qa.py <topic> [--llm]")
|
||||||
|
sys.exit(1)
|
||||||
|
asyncio.run(main(args[0], "--llm" in sys.argv))
|
||||||
@@ -30,6 +30,15 @@ class TopicCreateRequest(BaseModel):
|
|||||||
name: str = Field(min_length=1, max_length=100)
|
name: str = Field(min_length=1, max_length=100)
|
||||||
|
|
||||||
|
|
||||||
|
class QaRunRequest(BaseModel):
|
||||||
|
topic: str = Field(min_length=1, max_length=100)
|
||||||
|
llm: bool = True # wie das Gate: Echtheits-/Dubletten-Stichprobe inklusive
|
||||||
|
|
||||||
|
|
||||||
|
class RepairRequest(BaseModel):
|
||||||
|
topic: str = Field(min_length=1, max_length=100)
|
||||||
|
|
||||||
|
|
||||||
class BlocksCreateRequest(BaseModel):
|
class BlocksCreateRequest(BaseModel):
|
||||||
topic: str = Field(min_length=1, max_length=100)
|
topic: str = Field(min_length=1, max_length=100)
|
||||||
instructions: str = Field(default="", max_length=2000)
|
instructions: str = Field(default="", max_length=2000)
|
||||||
@@ -37,6 +46,7 @@ class BlocksCreateRequest(BaseModel):
|
|||||||
source_type: SourceType = "thema"
|
source_type: SourceType = "thema"
|
||||||
source_location: str = Field(default="", max_length=2000)
|
source_location: str = Field(default="", max_length=2000)
|
||||||
research: bool = True # False = Continue: drain the existing kanban queue, no new search
|
research: bool = True # False = Continue: drain the existing kanban queue, no new search
|
||||||
|
qa_force: bool = False # True = übersteuert ein pausierendes QA-Gate („Trotzdem fortsetzen")
|
||||||
|
|
||||||
|
|
||||||
class BlocksCardRestartRequest(BaseModel):
|
class BlocksCardRestartRequest(BaseModel):
|
||||||
|
|||||||
448
backend/qa.py
Normal file
448
backend/qa.py
Normal file
@@ -0,0 +1,448 @@
|
|||||||
|
"""Independent quality audit over a FINISHED generation run — read-only.
|
||||||
|
|
||||||
|
Measures the MECE goal ("no duplicates, no gaps") with detectors that deliberately
|
||||||
|
do NOT reuse the pipeline's heuristics (_canonical_key/_relation_conflict/_evidence_pack)
|
||||||
|
— shared blind spots would make the audit worthless. Shared infra only: DB access,
|
||||||
|
embedding.py, the agent runner (--llm sampling), atomic_write_json.
|
||||||
|
|
||||||
|
CLI: python3 qa.py <topic> [--llm] (or: make qa TOPIC=<topic> [LLM=1])
|
||||||
|
Report: storage/qa/<topic>/<run_id|timestamp>.json + console digest + diff to the
|
||||||
|
previous report of the same topic.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import database as db
|
||||||
|
import embedding
|
||||||
|
from config import STORAGE_DIR, SUB_DUP_KANDIDAT_COS
|
||||||
|
from fsutil import atomic_write_json
|
||||||
|
from jsonio import read_json_file as _json_file
|
||||||
|
from paths import arbeit_dir
|
||||||
|
from textkit import _norm_title
|
||||||
|
|
||||||
|
QA_DIR = STORAGE_DIR / "qa"
|
||||||
|
JACCARD_FLOOR = 0.5 # title token overlap that makes a pair suspicious
|
||||||
|
EMB_FLOOR = 0.82 # casefolded title cosine (own threshold, NOT the pipeline's 0.65)
|
||||||
|
SECTION_CHARS = 4000 # own paragraph splitter — independent of _text_sections
|
||||||
|
COVER_MIN_TOKENS = 2 # distinctive block tokens a section must share to count as covered
|
||||||
|
FREMD_MIN_TOKENS = 1 # distinctive title tokens that must appear in the corpus
|
||||||
|
LLM_SAMPLE = 12 # pairs/sections per judge call with --llm
|
||||||
|
# Note 0-10, deterministisch aus den Quoten (transparent, diffbar — keine LLM-"Gefühlsnote").
|
||||||
|
# Lücken/Fremd wiegen am schwersten (fehlender/falscher Stoff); Dubletten-VERDACHT enthält
|
||||||
|
# bewusst Rauschen und wiegt daher wenig.
|
||||||
|
NOTE_GEWICHTE = {"luecken": 3.0, "fremd": 2.5, "unechte_bloecke": 2.5, "hygiene": 0.5}
|
||||||
|
# subs/artefacts only exist after board 2 — at gate time these quotas would always be 0
|
||||||
|
# and water down the inventory score, hence a separate score.
|
||||||
|
# sub_dubletten counts only with --llm (confirmed pairs); the bare candidate list is
|
||||||
|
# suspicion (sub_dubletten_verdacht, weightless — like dubletten_verdacht).
|
||||||
|
NOTE_GEWICHTE_ARTEFAKTE = {"subs_ohne_beleg": 2.0, "verwaiste": 1.0, "sub_dubletten": 1.0}
|
||||||
|
|
||||||
|
_WORD = re.compile(r"\w{3,}")
|
||||||
|
_PAREN = re.compile(r"^\s*(.*?)\s*\(([^()]{2,60})\)\s*$")
|
||||||
|
_STOP = {"der", "die", "das", "und", "oder", "für", "mit", "von", "des", "den", "dem",
|
||||||
|
"ein", "eine", "the", "and", "for", "with", "als", "auf", "bei", "aus",
|
||||||
|
"problem", "algorithmus", "algorithm", "definition", "satz", "lemma"}
|
||||||
|
|
||||||
|
|
||||||
|
def _tokens(s: str) -> set[str]:
|
||||||
|
return {t for t in _WORD.findall((s or "").casefold()) if t not in _STOP}
|
||||||
|
|
||||||
|
|
||||||
|
def _distinctive(s: str) -> set[str]:
|
||||||
|
"""Tokens that can anchor a title in a corpus (stopword-free, ≥3 chars)."""
|
||||||
|
return _tokens(s)
|
||||||
|
|
||||||
|
|
||||||
|
def _ascii(t: str) -> str:
|
||||||
|
return "".join(c for c in t if c.isascii())
|
||||||
|
|
||||||
|
|
||||||
|
def _jaccard(a: set[str], b: set[str]) -> float:
|
||||||
|
return len(a & b) / len(a | b) if a | b else 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def _sections(text: str, goal: int | None = None) -> list[str]:
|
||||||
|
"""Own paragraph-boundary splitter (NOT blocks._text_sections — independence)."""
|
||||||
|
goal = goal or SECTION_CHARS
|
||||||
|
out, buf = [], ""
|
||||||
|
for para in re.split(r"\n\s*\n", text.strip()):
|
||||||
|
para = para.strip()
|
||||||
|
if not para:
|
||||||
|
continue
|
||||||
|
if buf and len(buf) + len(para) > goal:
|
||||||
|
out.append(buf)
|
||||||
|
buf = para
|
||||||
|
else:
|
||||||
|
buf = f"{buf}\n\n{para}" if buf else para
|
||||||
|
if buf.strip():
|
||||||
|
out.append(buf)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _corpus_texts(topic: str) -> dict[str, str]:
|
||||||
|
from blocks import source_folder # lazy: blocks pulls heavy deps
|
||||||
|
folder = source_folder(topic)
|
||||||
|
if not folder or not folder.is_dir():
|
||||||
|
return {}
|
||||||
|
out = {}
|
||||||
|
for f in sorted(folder.glob("*.txt")):
|
||||||
|
try:
|
||||||
|
out[f.name] = f.read_text(encoding="utf-8")
|
||||||
|
except OSError:
|
||||||
|
continue
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
# ── Detectors ───────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def dubletten(blocks: list[dict], emb_on: bool = True) -> list[dict]:
|
||||||
|
"""Suspicious pairs via signal UNION: token jaccard, casefolded-title embedding
|
||||||
|
cosine, paren acronym == other title. Every signal is independent of the pipeline."""
|
||||||
|
titles = [b["title"] for b in blocks]
|
||||||
|
toks = [_tokens(t) for t in titles]
|
||||||
|
sims = None
|
||||||
|
if emb_on and titles and embedding.available():
|
||||||
|
arr = embedding.embed([t.casefold() for t in titles])
|
||||||
|
if arr is not None:
|
||||||
|
sims = arr @ arr.T
|
||||||
|
ops = [bool(re.search(r"[≤⪯≥⊆⊊→⇒⟹⇔←]", t)) for t in titles]
|
||||||
|
out = []
|
||||||
|
for i in range(len(titles)):
|
||||||
|
for j in range(i + 1, len(titles)):
|
||||||
|
# relation vs. its operand ("Subset Sum" ⊂ "3-SAT ≤ Subset Sum"): by design
|
||||||
|
# separate entities — token containment there is expected, not suspicious
|
||||||
|
if ops[i] != ops[j] and (toks[i] <= toks[j] or toks[j] <= toks[i]):
|
||||||
|
continue
|
||||||
|
signals = {}
|
||||||
|
jac = _jaccard(toks[i], toks[j])
|
||||||
|
if jac >= JACCARD_FLOOR:
|
||||||
|
signals["jaccard"] = round(jac, 2)
|
||||||
|
if sims is not None and float(sims[i][j]) >= EMB_FLOOR:
|
||||||
|
signals["emb_cos"] = round(float(sims[i][j]), 2)
|
||||||
|
for a, b in ((i, j), (j, i)):
|
||||||
|
m = _PAREN.match(titles[a])
|
||||||
|
if m and _norm_title(titles[b]) in (_norm_title(m.group(1)), _norm_title(m.group(2))):
|
||||||
|
signals["akronym"] = True
|
||||||
|
if signals:
|
||||||
|
out.append({"a": titles[i], "b": titles[j], "signale": signals})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def sub_dubletten(sub_rows: list[dict], emb_on: bool = True) -> list[dict]:
|
||||||
|
"""Suspicious SUB pairs, in-block AND cross-block: casefolded title cosine ≥
|
||||||
|
SUB_DUP_KANDIDAT_COS. The pipeline's own merge paths act from 0.90 upward — the
|
||||||
|
measured bulk of real paraphrase duplicates sits in the band below, so everything
|
||||||
|
above the floor is a candidate. The verdict falls with --llm; without it this is
|
||||||
|
a suspicion list only (weightless)."""
|
||||||
|
cons = [r for r in sub_rows if r["status"] == "consensus"]
|
||||||
|
if len(cons) < 2 or not emb_on or not embedding.available():
|
||||||
|
return []
|
||||||
|
arr = embedding.embed([r["sub_title"].casefold() for r in cons])
|
||||||
|
if arr is None:
|
||||||
|
return []
|
||||||
|
sims = arr @ arr.T
|
||||||
|
out = []
|
||||||
|
for i in range(len(cons)):
|
||||||
|
for j in range(i + 1, len(cons)):
|
||||||
|
v = float(sims[i][j])
|
||||||
|
if v >= SUB_DUP_KANDIDAT_COS:
|
||||||
|
out.append({"a": f"[{cons[i]['block']}] {cons[i]['sub_title']}",
|
||||||
|
"b": f"[{cons[j]['block']}] {cons[j]['sub_title']}",
|
||||||
|
"cos": round(v, 2),
|
||||||
|
"cross": cons[i]["block_norm"] != cons[j]["block_norm"]})
|
||||||
|
return sorted(out, key=lambda p: -p["cos"])
|
||||||
|
|
||||||
|
|
||||||
|
def luecken(blocks: list[dict], subs_by_norm: dict[str, list[str]], corpus: dict[str, str]) -> list[dict]:
|
||||||
|
"""Corpus sections no block (title+description+subs tokens) sufficiently anchors.
|
||||||
|
Description tokens matter at the QA GATE: board 2 has not run yet, so titles alone
|
||||||
|
under-cover and inflate the quota."""
|
||||||
|
anchors: list[set[str]] = []
|
||||||
|
for b in blocks:
|
||||||
|
t = _distinctive(b["title"]) | _distinctive(b.get("description") or "")
|
||||||
|
for s in subs_by_norm.get(_norm_title(b["title"]), []):
|
||||||
|
t |= _distinctive(s)
|
||||||
|
anchors.append(t)
|
||||||
|
out = []
|
||||||
|
for fname, text in corpus.items():
|
||||||
|
for k, sec in enumerate(_sections(text), 1):
|
||||||
|
sec_toks = _tokens(sec)
|
||||||
|
covered = any(len(a & sec_toks) >= COVER_MIN_TOKENS for a in anchors)
|
||||||
|
if not covered:
|
||||||
|
preview = " ".join(sec.split())[:120]
|
||||||
|
out.append({"datei": fname, "abschnitt": k, "vorschau": preview})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def fremd(blocks: list[dict], corpus: dict[str, str]) -> list[str]:
|
||||||
|
"""Blocks whose distinctive title tokens never appear in the corpus (scope creep).
|
||||||
|
Token/stem match, NOT raw substring — 'bergang' ⊂ 'Übergang' had whitewashed the
|
||||||
|
garbage title 'αÜbergang'. The ASCII form only bridges symbol variants (Δ/∆)."""
|
||||||
|
ctoks = set(_WORD.findall("\n".join(corpus.values()).casefold()))
|
||||||
|
|
||||||
|
def _hit(t: str) -> bool:
|
||||||
|
forms = {t} | ({a} if len(a := _ascii(t)) >= 3 else set())
|
||||||
|
# digit-suffix fallback: '∆TSP1' → 'tsp1' misses the corpus token 'tsp' ('∆' is no \w)
|
||||||
|
forms |= {f2 for f in list(forms) if len(f2 := f.rstrip("0123456789")) >= 3}
|
||||||
|
return any(ct == f or ct.startswith(f) for f in forms for ct in ctoks)
|
||||||
|
|
||||||
|
out = []
|
||||||
|
for b in blocks:
|
||||||
|
dist = _distinctive(b["title"])
|
||||||
|
if dist and sum(1 for t in dist if _hit(t)) < FREMD_MIN_TOKENS:
|
||||||
|
out.append(b["title"])
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def beleg(blocks: list[dict], sub_rows: list[dict]) -> dict:
|
||||||
|
ohne_quelle = [b["title"] for b in blocks if not b.get("sources")]
|
||||||
|
ohne_mention = [f"{r['block']} · {r['sub_title']}" for r in sub_rows
|
||||||
|
if r["status"] != "variant" and not r["mentions"]]
|
||||||
|
return {"bloecke_ohne_quelle": ohne_quelle, "subs_ohne_beleg": ohne_mention}
|
||||||
|
|
||||||
|
|
||||||
|
def hygiene(blocks: list[dict]) -> list[dict]:
|
||||||
|
out = []
|
||||||
|
for b in blocks:
|
||||||
|
t = b["title"]
|
||||||
|
probleme = []
|
||||||
|
if "**" in t or "`" in t:
|
||||||
|
probleme.append("markdown")
|
||||||
|
if re.search(r"\(\d+\)\s*$", t):
|
||||||
|
probleme.append("kollisions-suffix")
|
||||||
|
if not (b.get("description") or "").strip():
|
||||||
|
probleme.append("leere-beschreibung")
|
||||||
|
if probleme:
|
||||||
|
out.append({"titel": t, "probleme": probleme})
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _zaehlbare_luecken(lk: list[dict], llm: bool) -> list[dict]:
|
||||||
|
"""With --llm only non-refuted gaps count ('?' = unjudged stays, conservative) — refuted
|
||||||
|
ones dragged the note although the judge cleared them (aak: 5 of 8, weight 3.0)."""
|
||||||
|
return [x for x in lk if x.get("llm") != "nein"] if llm else lk
|
||||||
|
|
||||||
|
|
||||||
|
def note(quoten: dict, gewichte: dict = NOTE_GEWICHTE) -> float:
|
||||||
|
"""10 = alle gewichteten Quoten 0. Gewicht = Punktabzug bei 100 % Quote (keine Normierung,
|
||||||
|
sonst staucht die Gewichtssumme die Skala nach oben). Ungemessene Quoten zählen nicht —
|
||||||
|
unechte_bloecke existiert nur mit --llm; dubletten_verdacht ist Verdachtsliste, kein Urteil."""
|
||||||
|
da = {k: w for k, w in gewichte.items() if k in quoten}
|
||||||
|
schaden = sum(w * min(float(quoten[k]), 1.0) for k, w in da.items())
|
||||||
|
return round(max(0.0, 10.0 * (1 - schaden)), 1)
|
||||||
|
|
||||||
|
|
||||||
|
def artefakte(sub_rows: list[dict], art_rows: list[dict], fragen: list[dict]) -> dict:
|
||||||
|
"""Coverage + orphans of the learning artefacts. Nenner = consensus-Subs (verworfene
|
||||||
|
zählen nicht als abzudeckendes Material). Waise = Ziel weder lebend (consensus/variant)
|
||||||
|
noch eindeutig als Kurztitel-Präfix von „kurztitel: beschreibung" auflösbar."""
|
||||||
|
if not art_rows and not fragen:
|
||||||
|
return {"status": "nicht generiert"}
|
||||||
|
cons = {(r["block_norm"], r["sub_norm"]) for r in sub_rows if r["status"] == "consensus"}
|
||||||
|
lebt = {(r["block_norm"], r["sub_norm"]) for r in sub_rows if r["status"] != "discarded"}
|
||||||
|
|
||||||
|
def _ziel(bn: str, sn: str):
|
||||||
|
if (bn, sn) in lebt:
|
||||||
|
return (bn, sn)
|
||||||
|
treffer = [k for k in lebt if k[0] == bn and k[1].startswith(sn + ":")]
|
||||||
|
if len(treffer) == 1:
|
||||||
|
return treffer[0]
|
||||||
|
# mehrere Treffer = meist ein consensus-Sub plus seine gefalteten Varianten
|
||||||
|
haupt = [k for k in treffer if k in cons]
|
||||||
|
return haupt[0] if len(haupt) == 1 else None
|
||||||
|
|
||||||
|
deck: dict[str, set] = {}
|
||||||
|
verwaist = []
|
||||||
|
for typ, bn, sn in ([(r["type"], r["block_norm"], r["sub_norm"]) for r in art_rows]
|
||||||
|
+ [("frage", r["block_norm"], r["sub_norm"]) for r in fragen]):
|
||||||
|
z = _ziel(bn, sn)
|
||||||
|
if z is None:
|
||||||
|
verwaist.append(f"{typ}: {bn} · {sn}")
|
||||||
|
else:
|
||||||
|
deck.setdefault(typ, set()).add(z)
|
||||||
|
n = max(len(cons), 1)
|
||||||
|
return {"status": "ok",
|
||||||
|
"frage_abdeckung": round(len(deck.get("frage", set()) & cons) / n, 3),
|
||||||
|
"flashcard_abdeckung": round(len(deck.get("flashcard", set()) & cons) / n, 3),
|
||||||
|
"beispiel_abdeckung": round(len(deck.get("example", set()) & cons) / n, 3),
|
||||||
|
"verwaiste": sorted(verwaist)}
|
||||||
|
|
||||||
|
|
||||||
|
# ── LLM sampling (optional, own prompts under templates/QA/) ────────────────────────
|
||||||
|
|
||||||
|
def _qa_prompt(name: str, **kwargs) -> str:
|
||||||
|
"""Own template dir (templates/QA/) — deliberately separate from the pipeline prompts."""
|
||||||
|
from config import TEMPLATES_DIR
|
||||||
|
return (TEMPLATES_DIR / "QA" / f"{name}.md").read_text(encoding="utf-8").format(**kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
async def _llm_verdicts(template: str, topic: str, key: str, items: list[str]) -> dict[int, str]:
|
||||||
|
from agents import run_agent
|
||||||
|
from pipeline import _yesno_schema
|
||||||
|
from jsonio import parse_json_text
|
||||||
|
listing = "\n\n".join(f"{k}. {it}" for k, it in enumerate(items, 1))
|
||||||
|
slot = {"Dubletten": "pairs", "Luecken": "sections", "Bausteine": "blocks", "Sub": "pairs"}[template.split("-")[1]]
|
||||||
|
rc, out, _err = await run_agent(f"qa-{topic}-{key}", _qa_prompt(template, topic=topic, extra="", **{slot: listing}),
|
||||||
|
600, role="judge", capabilities="none", scope=topic, label=f"QA {key}")
|
||||||
|
return (_yesno_schema(parse_json_text(out)) or {}) if rc == 0 else {}
|
||||||
|
|
||||||
|
|
||||||
|
# ── Report ──────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
async def qa_report(topic: str, llm: bool = False) -> dict | None:
|
||||||
|
cards = await db.kanban_cards(topic, board="inventory", stage="done_block")
|
||||||
|
if not cards:
|
||||||
|
print(f"Keine fertigen Blöcke für '{topic}' — Tippfehler im Namen oder Lauf nicht durch?")
|
||||||
|
return None
|
||||||
|
blocks = [{"title": c["payload"].get("title", ""), "description": c["payload"].get("description", ""),
|
||||||
|
"sources": c["payload"].get("sources") or []} for c in cards]
|
||||||
|
sub_rows = [dict(r) for bn in {_norm_title(b["title"]) for b in blocks}
|
||||||
|
for r in await db.list_subblocks(topic, bn)]
|
||||||
|
subs_by_norm: dict[str, list[str]] = {}
|
||||||
|
for r in sub_rows:
|
||||||
|
if r["status"] != "variant":
|
||||||
|
subs_by_norm.setdefault(r["block_norm"], []).append(r["sub_title"])
|
||||||
|
corpus = _corpus_texts(topic)
|
||||||
|
|
||||||
|
d = dubletten(blocks)
|
||||||
|
sd = sub_dubletten(sub_rows)
|
||||||
|
lk = luecken(blocks, subs_by_norm, corpus) if corpus else []
|
||||||
|
fr = fremd(blocks, corpus) if corpus else []
|
||||||
|
bl = beleg(blocks, sub_rows)
|
||||||
|
hy = hygiene(blocks)
|
||||||
|
n_sections = sum(len(_sections(t)) for t in corpus.values()) or 1
|
||||||
|
|
||||||
|
if llm and d:
|
||||||
|
v = await _llm_verdicts("QA-Dubletten", topic, "dubletten",
|
||||||
|
[f"A: {p['a']}\nB: {p['b']}" for p in d[:LLM_SAMPLE]])
|
||||||
|
for k, p in enumerate(d[:LLM_SAMPLE], 1):
|
||||||
|
p["llm"] = v.get(k, "?")
|
||||||
|
if llm and lk:
|
||||||
|
v = await _llm_verdicts("QA-Luecken", topic, "luecken",
|
||||||
|
[f"[{x['datei']} #{x['abschnitt']}] {x['vorschau']}" for x in lk[:LLM_SAMPLE]])
|
||||||
|
for k, x in enumerate(lk[:LLM_SAMPLE], 1):
|
||||||
|
x["llm"] = v.get(k, "?")
|
||||||
|
if llm and sd: # full coverage in chunks — a sampled quota would mislead the note
|
||||||
|
for lo in range(0, len(sd), 40):
|
||||||
|
chunk = sd[lo:lo + 40]
|
||||||
|
v = await _llm_verdicts("QA-Sub-Dubletten", topic, f"sub-dubletten-{lo}",
|
||||||
|
[f"A: {p['a']}\nB: {p['b']}" for p in chunk])
|
||||||
|
for k, p in enumerate(chunk, 1):
|
||||||
|
p["llm"] = v.get(k, "?")
|
||||||
|
unecht: list[str] | None = None
|
||||||
|
if llm and blocks:
|
||||||
|
unecht = []
|
||||||
|
for lo in range(0, len(blocks), 80): # ein Call je 80 Titel
|
||||||
|
chunk = blocks[lo:lo + 80]
|
||||||
|
v = await _llm_verdicts("QA-Bausteine", topic, f"bausteine-{lo}",
|
||||||
|
[f"{b['title']} — {b['description'] or '(ohne Beschreibung)'}" for b in chunk])
|
||||||
|
unecht += [b["title"] for k, b in enumerate(chunk, 1) if v.get(k) == "nein"]
|
||||||
|
|
||||||
|
art_rows = [dict(r) for r in await db.get_sub_artefakte(topic)]
|
||||||
|
fragen = [dict(r) for r in await db.list_question_pattern(topic)]
|
||||||
|
art = artefakte(sub_rows, art_rows, fragen)
|
||||||
|
quoten_art: dict[str, float] = {}
|
||||||
|
if sub_rows:
|
||||||
|
quoten_art["subs_ohne_beleg"] = round(len(bl["subs_ohne_beleg"]) / len(sub_rows), 3)
|
||||||
|
if art.get("status") == "ok":
|
||||||
|
quoten_art["verwaiste"] = round(len(art["verwaiste"]) / max(len(art_rows) + len(fragen), 1), 3)
|
||||||
|
n_cons = sum(1 for r in sub_rows if r["status"] == "consensus")
|
||||||
|
if n_cons:
|
||||||
|
quoten_art["sub_dubletten_verdacht"] = round(len(sd) / n_cons, 3)
|
||||||
|
if llm: # confirmed pairs only — the bare candidate list is suspicion, not damage
|
||||||
|
quoten_art["sub_dubletten"] = round(sum(1 for p in sd if p.get("llm") == "ja") / n_cons, 3)
|
||||||
|
summary = _json_file(arbeit_dir(topic) / "lauf-summary.json") or {}
|
||||||
|
report = {
|
||||||
|
"topic": topic, "erstellt": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"run_id": summary.get("run_id", ""), "bloecke": len(blocks),
|
||||||
|
"quoten": {
|
||||||
|
"dubletten_verdacht": round(len(d) / max(len(blocks), 1), 3),
|
||||||
|
"luecken": round(len(_zaehlbare_luecken(lk, llm)) / n_sections, 3),
|
||||||
|
"fremd": round(len(fr) / max(len(blocks), 1), 3),
|
||||||
|
"hygiene": round(len(hy) / max(len(blocks), 1), 3),
|
||||||
|
**({"unechte_bloecke": round(len(unecht) / max(len(blocks), 1), 3)} if unecht is not None else {}),
|
||||||
|
},
|
||||||
|
"quoten_artefakte": quoten_art,
|
||||||
|
**({"unecht": unecht} if unecht is not None else {}),
|
||||||
|
"dubletten": d, "sub_dubletten": sd, "luecken": lk, "fremd": fr, "beleg": bl, "hygiene": hy,
|
||||||
|
"artefakte": art,
|
||||||
|
"lauf": summary,
|
||||||
|
}
|
||||||
|
report["note"] = note(report["quoten"])
|
||||||
|
# None statt 10.0, solange Board 2 nichts geliefert hat — nichts gemessen ist keine Bestnote
|
||||||
|
report["note_artefakte"] = note(quoten_art, NOTE_GEWICHTE_ARTEFAKTE) if quoten_art else None
|
||||||
|
report["note_gewichte"] = {"inventar": NOTE_GEWICHTE, "artefakte": NOTE_GEWICHTE_ARTEFAKTE}
|
||||||
|
return report
|
||||||
|
|
||||||
|
|
||||||
|
def _diff(prev: dict | None, cur: dict) -> dict:
|
||||||
|
if not prev:
|
||||||
|
return {}
|
||||||
|
# ältere Reports führten die Artefakt-Quoten noch unter "quoten"
|
||||||
|
alt = {**prev.get("quoten", {}), **prev.get("quoten_artefakte", {})}
|
||||||
|
neu = {**cur["quoten"], **cur.get("quoten_artefakte", {})}
|
||||||
|
return {k: round(v - alt.get(k, 0), 3) for k, v in neu.items()}
|
||||||
|
|
||||||
|
|
||||||
|
def _write_report(report: dict) -> Path:
|
||||||
|
tdir = QA_DIR / report["topic"]
|
||||||
|
tdir.mkdir(parents=True, exist_ok=True)
|
||||||
|
# by mtime: run-id names (…-1311-5e5c) and timestamp names don't sort lexicographically.
|
||||||
|
# guide-* reports share the directory but are a SEPARATE series (guide_qa.py).
|
||||||
|
older = sorted((p for p in tdir.glob("*.json") if not p.name.startswith("guide-")),
|
||||||
|
key=lambda p: p.stat().st_mtime)
|
||||||
|
prev = _json_file(older[-1]) if older else None
|
||||||
|
report["diff_zum_vorlauf"] = _diff(prev, report)
|
||||||
|
name = report["run_id"] or datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
|
||||||
|
path = tdir / f"{name}.json"
|
||||||
|
atomic_write_json(path, report, indent=1)
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
def _digest(report: dict, path: Path):
|
||||||
|
na = report.get("note_artefakte")
|
||||||
|
print(f"QA {report['topic']} — {report['bloecke']} Blöcke (run {report['run_id'] or '—'})"
|
||||||
|
f" — Inventar {report['note']}/10 · Artefakte {f'{na}/10' if na is not None else '—'}")
|
||||||
|
for k, v in {**report["quoten"], **report.get("quoten_artefakte", {})}.items():
|
||||||
|
delta = report.get("diff_zum_vorlauf", {}).get(k)
|
||||||
|
d = f" ({'+' if delta > 0 else ''}{delta})" if delta else ""
|
||||||
|
print(f" {k:20} {v:6.1%}{d}")
|
||||||
|
for p in report["dubletten"][:8]:
|
||||||
|
print(f" DUBLETTE? {p['a']} <-> {p['b']} {p['signale']}{' LLM:' + p['llm'] if 'llm' in p else ''}")
|
||||||
|
for p in report.get("sub_dubletten", [])[:8]:
|
||||||
|
print(f" SUB-DUP? {p['a']} <-> {p['b']} cos={p['cos']}{' LLM:' + p['llm'] if 'llm' in p else ''}")
|
||||||
|
for t in report["fremd"][:8]:
|
||||||
|
print(f" FREMD? {t}")
|
||||||
|
for t in report.get("unecht", [])[:8]:
|
||||||
|
print(f" UNECHT {t}")
|
||||||
|
art = report["artefakte"]
|
||||||
|
if art.get("status") == "ok":
|
||||||
|
print(f" Artefakte: Frage {art['frage_abdeckung']:.0%} · Flashcard {art['flashcard_abdeckung']:.0%}"
|
||||||
|
f" · Beispiel {art['beispiel_abdeckung']:.0%} · verwaist {len(art['verwaiste'])}")
|
||||||
|
else:
|
||||||
|
print(" Artefakte: nicht generiert (Board 2 nicht gelaufen)")
|
||||||
|
print(f"Report: {path}")
|
||||||
|
|
||||||
|
|
||||||
|
async def main(topic: str, llm: bool):
|
||||||
|
await db.init_db()
|
||||||
|
try:
|
||||||
|
report = await qa_report(topic, llm=llm)
|
||||||
|
if report is None:
|
||||||
|
sys.exit(1)
|
||||||
|
_digest(report, _write_report(report))
|
||||||
|
finally:
|
||||||
|
await db.close_db()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||||
|
if not args:
|
||||||
|
print("Nutzung: python3 qa.py <topic> [--llm]")
|
||||||
|
sys.exit(1)
|
||||||
|
asyncio.run(main(args[0], "--llm" in sys.argv))
|
||||||
325
backend/repair.py
Normal file
325
backend/repair.py
Normal file
@@ -0,0 +1,325 @@
|
|||||||
|
"""Befund-Repair: arbeitet den jüngsten QA-Report gezielt ab — ohne Flow, ohne Board-Rebuild.
|
||||||
|
|
||||||
|
Blindes Re-Filtern reproduziert die blinden Flecken der Pipeline (sie hat die Befunde ja
|
||||||
|
durchgelassen). Hier fließen die QA-BEFUNDE als Input in gezielte Aktionen: Hygiene
|
||||||
|
deterministisch, bestätigte Dubletten mergen (Zweitmeinung), Fremd/Unecht nur nach
|
||||||
|
Gegen-Judge entfernen (fail-open: Zweifel/Fehler → behalten). Lücken brauchen Recherche,
|
||||||
|
Verwaiste den nächsten Board-2-Lauf — beides wird nur ausgewiesen."""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import json
|
||||||
|
import logging
|
||||||
|
import re
|
||||||
|
|
||||||
|
import database as db
|
||||||
|
import qa
|
||||||
|
from agents import run_agent
|
||||||
|
from blocks import _blocks_files, _evidence_pack, source_folder
|
||||||
|
from fsutil import atomic_write_json
|
||||||
|
from jsonio import parse_json_text, read_json_file as _json_file
|
||||||
|
from pipeline import _yesno_schema
|
||||||
|
from textkit import _norm_title, _title, clean_title
|
||||||
|
|
||||||
|
log = logging.getLogger("creator.repair")
|
||||||
|
|
||||||
|
JUDGE_TIMEOUT = 600
|
||||||
|
JUDGE_CHUNK = 40 # Befunde je Judge-Call
|
||||||
|
EVIDENCE_PER_BLOCK = 6000 # Zeichen Material-Auszug je Fremd-Kandidat
|
||||||
|
|
||||||
|
|
||||||
|
async def repair_befunde(topic: str) -> dict:
|
||||||
|
tdir = qa.QA_DIR / topic
|
||||||
|
reports = sorted((p for p in tdir.glob("*.json") if not p.name.startswith("guide-")),
|
||||||
|
key=lambda p: p.stat().st_mtime) if tdir.is_dir() else []
|
||||||
|
report = _json_file(reports[-1]) if reports else None
|
||||||
|
if not report:
|
||||||
|
return {"fehler": "kein QA-Report — erst QA laufen lassen"}
|
||||||
|
files = _blocks_files(topic)
|
||||||
|
cards = await db.kanban_cards(topic, board="inventory", stage="done_block")
|
||||||
|
by_norm = {_norm_title(c["payload"].get("title", "")): c for c in cards}
|
||||||
|
|
||||||
|
hygiene = await _fix_hygiene(topic, report, by_norm, files)
|
||||||
|
merges = await _merge_dubletten(topic, report, by_norm, files)
|
||||||
|
sub_merges = await _merge_sub_dubletten(topic, report, files)
|
||||||
|
entfernt = await _entferne_fremd_unecht(topic, report, by_norm, files)
|
||||||
|
aufgeraeumt = await _raeume_waisen(topic)
|
||||||
|
|
||||||
|
# llm=True: gleiche Messlatte wie QA-Button/Abschluss-QA — der llm=False-Report
|
||||||
|
# blendete sub_dubletten aus und ließ die Note zwischen 10.0 und ~9 pendeln
|
||||||
|
neu = await qa.qa_report(topic, llm=True)
|
||||||
|
if neu:
|
||||||
|
await asyncio.to_thread(qa._write_report, neu)
|
||||||
|
return {"hygiene": hygiene, "merges": merges, "sub_merges": sub_merges, "entfernt": entfernt,
|
||||||
|
"aufgeraeumt": aufgeraeumt, "braucht_research": len(report.get("luecken", []))}
|
||||||
|
|
||||||
|
|
||||||
|
async def _judge(template: str, topic: str, key: str, slot: str, items: list[str]) -> dict[int, str]:
|
||||||
|
"""No-Tool-Judge-Wellen über alle Items (fail-open: Fehler → leeres Verdikt = behalten)."""
|
||||||
|
verdicts: dict[int, str] = {}
|
||||||
|
for lo in range(0, len(items), JUDGE_CHUNK):
|
||||||
|
chunk = items[lo:lo + JUDGE_CHUNK]
|
||||||
|
listing = "\n\n".join(f"{k}. {it}" for k, it in enumerate(chunk, 1))
|
||||||
|
try:
|
||||||
|
rc, out, _err = await run_agent(
|
||||||
|
f"repair-{topic}-{key}-{lo}", qa._qa_prompt(template, topic=topic, extra="", **{slot: listing}),
|
||||||
|
JUDGE_TIMEOUT, role="judge", capabilities="none", scope=topic, label=f"Repair {key}")
|
||||||
|
v = (_yesno_schema(parse_json_text(out)) or {}) if rc == 0 else {}
|
||||||
|
except Exception:
|
||||||
|
log.exception("[%s] Repair-Judge %s fehlgeschlagen — Befunde bleiben", topic, key)
|
||||||
|
v = {}
|
||||||
|
verdicts.update({lo + k: urteil for k, urteil in v.items()})
|
||||||
|
return verdicts
|
||||||
|
|
||||||
|
|
||||||
|
async def _fix_hygiene(topic: str, report: dict, by_norm: dict, files: dict) -> list[str]:
|
||||||
|
"""Nur der norm-invariante Teil (`**`/Backticks); `(n)`-Suffix und leere Beschreibung
|
||||||
|
ändern die Norm bzw. brauchen Inhalt — bleiben Befund."""
|
||||||
|
fixed = []
|
||||||
|
for h in report.get("hygiene", []):
|
||||||
|
alt = h.get("titel", "")
|
||||||
|
neu = clean_title(alt)
|
||||||
|
if neu == alt or _norm_title(neu) != _norm_title(alt):
|
||||||
|
continue
|
||||||
|
norm = _norm_title(alt)
|
||||||
|
card = by_norm.get(norm)
|
||||||
|
if not card:
|
||||||
|
continue
|
||||||
|
p = dict(card["payload"])
|
||||||
|
p["title"] = neu
|
||||||
|
await db.kanban_set_payload(topic, "inventory", card["card_id"], p)
|
||||||
|
await db.set_block_status(topic, norm, "consensus", title=neu)
|
||||||
|
_rename_in_files(files, norm, neu)
|
||||||
|
fixed.append(f"{alt} → {neu}")
|
||||||
|
return fixed
|
||||||
|
|
||||||
|
|
||||||
|
async def _merge_dubletten(topic: str, report: dict, by_norm: dict, files: dict) -> list[str]:
|
||||||
|
"""Nur QA-bestätigte Paare (llm=ja); eine Zweitmeinung, Merge nur bei erneut ja.
|
||||||
|
Merge spiegelt die dedup-Stage: Union ins Gewinner-Payload, Verlierer → grouped."""
|
||||||
|
paare = [p for p in report.get("dubletten", []) if p.get("llm") == "ja"
|
||||||
|
and _norm_title(p.get("a", "")) in by_norm and _norm_title(p.get("b", "")) in by_norm]
|
||||||
|
if not paare:
|
||||||
|
return []
|
||||||
|
v = await _judge("QA-Dubletten", topic, "dubletten", "pairs",
|
||||||
|
[f"A: {p['a']}\nB: {p['b']}" for p in paare])
|
||||||
|
merged = []
|
||||||
|
for i, p in enumerate(paare, 1):
|
||||||
|
a, b = by_norm.get(_norm_title(p["a"])), by_norm.get(_norm_title(p["b"]))
|
||||||
|
if v.get(i) != "ja" or not a or not b or a["card_id"] == b["card_id"]:
|
||||||
|
continue
|
||||||
|
win, lose = sorted((a, b), key=lambda c: (len(c["payload"].get("description") or ""),
|
||||||
|
len(c["payload"].get("title") or "")), reverse=True)
|
||||||
|
wp, lp = dict(win["payload"]), dict(lose["payload"])
|
||||||
|
wp["readers"] = sorted(set(wp.get("readers") or []) | set(lp.get("readers") or []))
|
||||||
|
wp["sources"] = sorted(set(wp.get("sources") or []) | set(lp.get("sources") or []))
|
||||||
|
lp.update(reason="merged", merged_into=wp.get("title", ""))
|
||||||
|
await db.kanban_set_payload(topic, "inventory", win["card_id"], wp)
|
||||||
|
await db.kanban_set_payload(topic, "inventory", lose["card_id"], lp)
|
||||||
|
await db.kanban_advance(topic, "inventory", lose["card_id"], "grouped")
|
||||||
|
await _purge_block(topic, lp.get("title", ""), files)
|
||||||
|
by_norm.pop(_norm_title(lp.get("title", "")), None)
|
||||||
|
merged.append(f"{lp.get('title')} → {wp.get('title')}")
|
||||||
|
return merged
|
||||||
|
|
||||||
|
|
||||||
|
_SUB_PAAR = re.compile(r"^\[(.+?)\] (.+)$", re.S)
|
||||||
|
|
||||||
|
|
||||||
|
def _sub_gewinner(a: dict, b: dict) -> tuple[dict, dict]:
|
||||||
|
"""Gewinner = mehr key_points im facts-Feld, dann längerer Titel (Muster Konsolidierung)."""
|
||||||
|
def score(r):
|
||||||
|
try:
|
||||||
|
kp = len((json.loads(r.get("facts") or "{}")).get("key_points") or [])
|
||||||
|
except ValueError:
|
||||||
|
kp = 0
|
||||||
|
return (kp, len(r.get("sub_title") or ""))
|
||||||
|
return (a, b) if score(a) >= score(b) else (b, a)
|
||||||
|
|
||||||
|
|
||||||
|
async def _merge_sub_dubletten(topic: str, report: dict, files: dict) -> list[str]:
|
||||||
|
"""QA-bestätigte Sub-Paare (llm=ja) nach Zweitmeinung falten: Verlierer → variant,
|
||||||
|
seine Fragen/Artefakte wandern zum Gewinner (oder fallen weg, wenn er den Typ hat).
|
||||||
|
Repair hatte dafür keinen Handler — die Paare überlebten jeden Repair-Zyklus."""
|
||||||
|
rows = {(r["block_norm"], r["sub_norm"]): r for r in await db.list_subblocks(topic)
|
||||||
|
if r["status"] == "consensus"}
|
||||||
|
|
||||||
|
def _row(eintrag: str):
|
||||||
|
m = _SUB_PAAR.match(eintrag or "")
|
||||||
|
return rows.get((_norm_title(m.group(1)), _norm_title(m.group(2)))) if m else None
|
||||||
|
|
||||||
|
paare = [(a, b) for p in report.get("sub_dubletten", []) if p.get("llm") == "ja"
|
||||||
|
and (a := _row(p.get("a"))) and (b := _row(p.get("b")))
|
||||||
|
and (a["block_norm"], a["sub_norm"]) != (b["block_norm"], b["sub_norm"])]
|
||||||
|
if not paare:
|
||||||
|
return []
|
||||||
|
v = await _judge("QA-Sub-Dubletten", topic, "sub-dubletten", "pairs",
|
||||||
|
[f"A: [{a['block']}] {a['sub_title']}\nB: [{b['block']}] {b['sub_title']}"
|
||||||
|
for a, b in paare])
|
||||||
|
merged: list[str] = []
|
||||||
|
gone: set[tuple] = set()
|
||||||
|
for i, (a, b) in enumerate(paare, 1):
|
||||||
|
win, lose = _sub_gewinner(a, b)
|
||||||
|
wk, lk = (win["block_norm"], win["sub_norm"]), (lose["block_norm"], lose["sub_norm"])
|
||||||
|
if v.get(i) != "ja" or wk in gone or lk in gone:
|
||||||
|
continue
|
||||||
|
await db.set_subblock_fields(topic, lose["block_norm"], lose["sub_norm"], status="variant")
|
||||||
|
gone.add(lk)
|
||||||
|
# Fragen/Artefakte des Verlierers: umhängen, wenn der Gewinner den Typ nicht hat
|
||||||
|
w_fragen = {r["sub_norm"] for r in await db.list_question_pattern(topic, win["block_norm"])}
|
||||||
|
for r in await db.list_question_pattern(topic, lose["block_norm"]):
|
||||||
|
if r["sub_norm"] != lose["sub_norm"]:
|
||||||
|
continue
|
||||||
|
if win["sub_norm"] not in w_fragen:
|
||||||
|
await db.upsert_question_pattern(topic, win["block_norm"], win["sub_norm"],
|
||||||
|
win["block"], win["sub_title"], r["question"])
|
||||||
|
await db.delete_frage_row(topic, lose["block_norm"], lose["sub_norm"])
|
||||||
|
w_typen = {r["type"] for r in await db.get_sub_artefakte(topic, block_norm=win["block_norm"])
|
||||||
|
if r["sub_norm"] == win["sub_norm"]}
|
||||||
|
for r in await db.get_sub_artefakte(topic, block_norm=lose["block_norm"]):
|
||||||
|
if r["sub_norm"] != lose["sub_norm"]:
|
||||||
|
continue
|
||||||
|
if r["type"] not in w_typen:
|
||||||
|
await db.put_sub_artifact(topic, win["block_norm"], win["sub_norm"], r["type"],
|
||||||
|
r["data"], win["block"], win["sub_title"])
|
||||||
|
await db.delete_artefakt_row(topic, lose["block_norm"], lose["sub_norm"], r["type"])
|
||||||
|
_entferne_sub_in_files(files, lose["block_norm"], lose["sub_norm"])
|
||||||
|
merged.append(f"{lose['sub_title'][:40]} → {win['sub_title'][:40]}")
|
||||||
|
return merged
|
||||||
|
|
||||||
|
|
||||||
|
def _entferne_sub_in_files(files: dict, bnorm: str, sub_norm: str) -> None:
|
||||||
|
"""Verlierer-Sub aus den Sidecar-JSONs nehmen (Legacy-Lesepfad von Guide/Frontend);
|
||||||
|
die DB trägt die umgehängten Fragen/Artefakte."""
|
||||||
|
for key, feld in (("sidecar", "title"), ("sub_roh", None), ("question_pattern", "subblock")):
|
||||||
|
d = _json_file(files[key])
|
||||||
|
if not isinstance(d, dict):
|
||||||
|
continue
|
||||||
|
changed = False
|
||||||
|
for bt, eintraege in d.items():
|
||||||
|
if _norm_title(bt) != bnorm or not isinstance(eintraege, list):
|
||||||
|
continue
|
||||||
|
neu = [e for e in eintraege
|
||||||
|
if _norm_title(e if feld is None else str((e or {}).get(feld, ""))) != sub_norm]
|
||||||
|
if len(neu) != len(eintraege):
|
||||||
|
d[bt] = neu
|
||||||
|
changed = True
|
||||||
|
if changed:
|
||||||
|
atomic_write_json(files[key], d, indent=1)
|
||||||
|
art = _json_file(files["artefakte"])
|
||||||
|
if isinstance(art, dict):
|
||||||
|
neu = {t: [e for e in (es if isinstance(es, list) else [])
|
||||||
|
if not (_norm_title(_title(str(e.get("block", "")))) == bnorm
|
||||||
|
and _norm_title(str(e.get("subblock", ""))) == sub_norm)]
|
||||||
|
for t, es in art.items()}
|
||||||
|
if neu != art:
|
||||||
|
atomic_write_json(files["artefakte"], neu, indent=1)
|
||||||
|
|
||||||
|
|
||||||
|
async def _entferne_fremd_unecht(topic: str, report: dict, by_norm: dict, files: dict) -> list[str]:
|
||||||
|
out = []
|
||||||
|
fremd = [t for t in report.get("fremd", []) if _norm_title(t) in by_norm]
|
||||||
|
if fremd:
|
||||||
|
folder = source_folder(topic)
|
||||||
|
lines = []
|
||||||
|
for t in fremd:
|
||||||
|
srcs = by_norm[_norm_title(t)]["payload"].get("sources") or None
|
||||||
|
ev = _evidence_pack(folder, srcs, [t], budget=EVIDENCE_PER_BLOCK) if folder else ""
|
||||||
|
lines.append(f"{t}\n{ev or '(keine Treffer im Material)'}")
|
||||||
|
v = await _judge("QA-Repair-Beleg", topic, "fremd", "blocks", lines)
|
||||||
|
for i, t in enumerate(fremd, 1):
|
||||||
|
if v.get(i) == "nein":
|
||||||
|
await _reject(topic, t, by_norm, files, "qa-fremd")
|
||||||
|
out.append(t)
|
||||||
|
unecht = [t for t in report.get("unecht", []) if _norm_title(t) in by_norm]
|
||||||
|
if unecht:
|
||||||
|
lines = [f"{t} — {by_norm[_norm_title(t)]['payload'].get('description') or '(ohne Beschreibung)'}"
|
||||||
|
for t in unecht]
|
||||||
|
v = await _judge("QA-Bausteine", topic, "unecht", "blocks", lines)
|
||||||
|
for i, t in enumerate(unecht, 1):
|
||||||
|
if v.get(i) == "nein":
|
||||||
|
await _reject(topic, t, by_norm, files, "qa-unecht")
|
||||||
|
out.append(t)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
async def _raeume_waisen(topic: str) -> int:
|
||||||
|
"""Artefakte/Fragen mit totem Ziel löschen (Sub verworfen oder weg) — inert, der
|
||||||
|
Übungs-Join spielt sie nie aus, aber sie drücken die Artefakt-Note. Mehrdeutige
|
||||||
|
Präfix-Treffer bleiben (könnten lebend sein — Löschen wäre riskanter als behalten)."""
|
||||||
|
lebt = {(r["block_norm"], r["sub_norm"]) for r in await db.list_subblocks(topic)
|
||||||
|
if r["status"] != "discarded"}
|
||||||
|
|
||||||
|
def tot(bn: str, sn: str) -> bool:
|
||||||
|
if (bn, sn) in lebt:
|
||||||
|
return False
|
||||||
|
return not any(b == bn and s.startswith(sn + ":") for b, s in lebt)
|
||||||
|
|
||||||
|
n = 0
|
||||||
|
for r in await db.get_sub_artefakte(topic):
|
||||||
|
if tot(r["block_norm"], r["sub_norm"]):
|
||||||
|
await db.delete_artefakt_row(topic, r["block_norm"], r["sub_norm"], r["type"])
|
||||||
|
n += 1
|
||||||
|
for r in await db.list_question_pattern(topic):
|
||||||
|
if tot(r["block_norm"], r["sub_norm"]):
|
||||||
|
await db.delete_frage_row(topic, r["block_norm"], r["sub_norm"])
|
||||||
|
n += 1
|
||||||
|
return n
|
||||||
|
|
||||||
|
|
||||||
|
async def _reject(topic: str, title: str, by_norm: dict, files: dict, grund: str) -> None:
|
||||||
|
norm = _norm_title(title)
|
||||||
|
card = by_norm.pop(norm, None)
|
||||||
|
if not card:
|
||||||
|
return
|
||||||
|
p = dict(card["payload"])
|
||||||
|
p["reason"] = grund
|
||||||
|
await db.kanban_set_payload(topic, "inventory", card["card_id"], p)
|
||||||
|
await db.kanban_advance(topic, "inventory", card["card_id"], "rejected")
|
||||||
|
await _purge_block(topic, title, files)
|
||||||
|
|
||||||
|
|
||||||
|
async def _purge_block(topic: str, title: str, files: dict) -> None:
|
||||||
|
"""Abgeleitete Daten eines Blocks gezielt entfernen (DB-Spiegel, Board-2-Karte, Sidecars)."""
|
||||||
|
norm = _norm_title(title)
|
||||||
|
await db.set_block_status(topic, norm, "discarded")
|
||||||
|
await db.delete_subblocks(topic, norm)
|
||||||
|
await db.delete_question_pattern(topic, norm)
|
||||||
|
await db.delete_sub_artefakte(topic, norm)
|
||||||
|
await db.kanban_delete_card(topic, "artefacts", norm)
|
||||||
|
for key in ("sidecar", "facts", "question_pattern", "sub_roh"):
|
||||||
|
d = _json_file(files[key])
|
||||||
|
if isinstance(d, dict):
|
||||||
|
hits = [k for k in d if _norm_title(k) == norm]
|
||||||
|
if hits:
|
||||||
|
for k in hits:
|
||||||
|
d.pop(k)
|
||||||
|
atomic_write_json(files[key], d, indent=1)
|
||||||
|
art = _json_file(files["artefakte"])
|
||||||
|
if isinstance(art, dict):
|
||||||
|
neu = {t: [e for e in (es if isinstance(es, list) else [])
|
||||||
|
if _norm_title(_title(str(e.get("block", "")))) != norm]
|
||||||
|
for t, es in art.items()}
|
||||||
|
if neu != art:
|
||||||
|
atomic_write_json(files["artefakte"], neu, indent=1)
|
||||||
|
|
||||||
|
|
||||||
|
def _rename_in_files(files: dict, norm: str, neu: str) -> None:
|
||||||
|
"""Titel-Keys der Sidecar-JSONs + artefakte-`block`-Felder auf den bereinigten Titel."""
|
||||||
|
for key in ("sidecar", "facts", "question_pattern", "sub_roh"):
|
||||||
|
d = _json_file(files[key])
|
||||||
|
if isinstance(d, dict):
|
||||||
|
hits = [k for k in d if _norm_title(k) == norm and k != neu]
|
||||||
|
if hits:
|
||||||
|
for k in hits:
|
||||||
|
d[neu] = d.pop(k)
|
||||||
|
atomic_write_json(files[key], d, indent=1)
|
||||||
|
art = _json_file(files["artefakte"])
|
||||||
|
if isinstance(art, dict):
|
||||||
|
changed = False
|
||||||
|
for es in art.values():
|
||||||
|
for e in es if isinstance(es, list) else []:
|
||||||
|
if _norm_title(_title(str(e.get("block", "")))) == norm and e.get("block") != neu:
|
||||||
|
e["block"] = neu
|
||||||
|
changed = True
|
||||||
|
if changed:
|
||||||
|
atomic_write_json(files["artefakte"], art, indent=1)
|
||||||
@@ -29,7 +29,7 @@ from rules import FORMATE, formats_stats, guide_lock, ist_completed, load_learns
|
|||||||
from models import (
|
from models import (
|
||||||
GuideCreateRequest, GuideResponse,
|
GuideCreateRequest, GuideResponse,
|
||||||
TopicCreateRequest,
|
TopicCreateRequest,
|
||||||
BlocksCreateRequest, BlocksResetStageRequest, BlocksCardRestartRequest, BlocksStatusResponse,
|
BlocksCreateRequest, BlocksResetStageRequest, BlocksCardRestartRequest, BlocksStatusResponse, QaRunRequest, RepairRequest,
|
||||||
GuideCardResetRequest, GuideFormatRequest,
|
GuideCardResetRequest, GuideFormatRequest,
|
||||||
GuideBoardResetRequest, GuideChatRequest, GuideChatResponse,
|
GuideBoardResetRequest, GuideChatRequest, GuideChatResponse,
|
||||||
ProviderInfo,
|
ProviderInfo,
|
||||||
@@ -140,7 +140,8 @@ async def create_blocks(req: BlocksCreateRequest):
|
|||||||
raise HTTPException(400, "Link must start with http:// or https://.")
|
raise HTTPException(400, "Link must start with http:// or https://.")
|
||||||
qp.parent.mkdir(parents=True, exist_ok=True)
|
qp.parent.mkdir(parents=True, exist_ok=True)
|
||||||
atomic_write_json(qp, {"type": type, "location": location, "spec": req.instructions.strip()})
|
atomic_write_json(qp, {"type": type, "location": location, "spec": req.instructions.strip()})
|
||||||
asyncio.create_task(generate_blocks(topic, req.instructions.strip(), req.provider, research=req.research))
|
asyncio.create_task(generate_blocks(topic, req.instructions.strip(), req.provider,
|
||||||
|
research=req.research, qa_force=req.qa_force))
|
||||||
return {"ok": True}
|
return {"ok": True}
|
||||||
|
|
||||||
|
|
||||||
@@ -157,6 +158,48 @@ async def get_blocks_board(topic: str):
|
|||||||
return snap
|
return snap
|
||||||
|
|
||||||
|
|
||||||
|
_qa_laeuft: set[str] = set()
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/blocks/qa")
|
||||||
|
async def run_qa_route(req: QaRunRequest):
|
||||||
|
"""Manual QA run (like the gate: incl. LLM samples); the badge reads the written report."""
|
||||||
|
if req.topic in _qa_laeuft:
|
||||||
|
return {"status": "läuft bereits"}
|
||||||
|
_qa_laeuft.add(req.topic)
|
||||||
|
try:
|
||||||
|
import qa
|
||||||
|
report = await qa.qa_report(req.topic, llm=req.llm)
|
||||||
|
if report is None:
|
||||||
|
raise HTTPException(status_code=404, detail="keine fertigen Bausteine")
|
||||||
|
await asyncio.to_thread(qa._write_report, report)
|
||||||
|
return {"note": report["note"], "note_artefakte": report["note_artefakte"]}
|
||||||
|
finally:
|
||||||
|
_qa_laeuft.discard(req.topic)
|
||||||
|
|
||||||
|
|
||||||
|
_repair_laeuft: set[str] = set()
|
||||||
|
|
||||||
|
|
||||||
|
@router.post("/blocks/repair")
|
||||||
|
async def run_repair_route(req: RepairRequest):
|
||||||
|
"""Fix the latest QA findings in place: hygiene, confirmed duplicates, foreign/unreal blocks."""
|
||||||
|
topic = req.topic.strip()
|
||||||
|
if (await blocks_status(topic))["generating"]:
|
||||||
|
return {"status": "generating"}
|
||||||
|
if topic in _repair_laeuft:
|
||||||
|
return {"status": "läuft bereits"}
|
||||||
|
_repair_laeuft.add(topic)
|
||||||
|
try:
|
||||||
|
import repair
|
||||||
|
res = await repair.repair_befunde(topic)
|
||||||
|
if "fehler" in res:
|
||||||
|
raise HTTPException(status_code=404, detail=res["fehler"])
|
||||||
|
return res
|
||||||
|
finally:
|
||||||
|
_repair_laeuft.discard(topic)
|
||||||
|
|
||||||
|
|
||||||
@router.post("/blocks/research")
|
@router.post("/blocks/research")
|
||||||
async def add_blocks_research(topic: str, provider: str = "claude"):
|
async def add_blocks_research(topic: str, provider: str = "claude"):
|
||||||
"""Attach one more research agent — to the live flow, or attach-or-start."""
|
"""Attach one more research agent — to the live flow, or attach-or-start."""
|
||||||
|
|||||||
@@ -56,6 +56,16 @@ async def board_env(testdb, tmp_path, monkeypatch):
|
|||||||
|
|
||||||
monkeypatch.setattr(bi, "run_single_slot", _fake_single_slot(tmp_path))
|
monkeypatch.setattr(bi, "run_single_slot", _fake_single_slot(tmp_path))
|
||||||
|
|
||||||
|
# QA-Gate: standardmäßig saubere Fake-Note (kein Embedding-Load in Tests);
|
||||||
|
# Gate-Tests überschreiben qa_report gezielt.
|
||||||
|
import qa as qa_mod
|
||||||
|
|
||||||
|
async def _fake_qa(topic, llm=False):
|
||||||
|
return {"note": 10.0, "topic": TOPIC, "quoten": {}, "fremd": [], "artefakte": {"status": "nicht generiert"}}
|
||||||
|
monkeypatch.setattr(qa_mod, "qa_report", _fake_qa)
|
||||||
|
monkeypatch.setattr(qa_mod, "_write_report", lambda r: None)
|
||||||
|
monkeypatch.setattr(bi, "_QA_GATE_POLL", 0.05)
|
||||||
|
|
||||||
async def no_emb(flow):
|
async def no_emb(flow):
|
||||||
return False
|
return False
|
||||||
monkeypatch.setattr(bi, "_emb_ok", no_emb)
|
monkeypatch.setattr(bi, "_emb_ok", no_emb)
|
||||||
@@ -86,12 +96,21 @@ async def board_env(testdb, tmp_path, monkeypatch):
|
|||||||
async def fake_outline(ctx, set_p, files, entries, instructions):
|
async def fake_outline(ctx, set_p, files, entries, instructions):
|
||||||
return {"chapters": [{"title": "Kapitel 1", "numbers": sorted(entries)}]}
|
return {"chapters": [{"title": "Kapitel 1", "numbers": sorted(entries)}]}
|
||||||
|
|
||||||
|
async def fake_konsolidierung(ctx, files, raw, facts_map, instructions="", ns="", lbl=""):
|
||||||
|
return None
|
||||||
|
|
||||||
for name, fn in [("_subblocks_block", fake_subblocks), ("_facts_block", fake_facts),
|
for name, fn in [("_subblocks_block", fake_subblocks), ("_facts_block", fake_facts),
|
||||||
("_levels_block", fake_levels), ("_relevance_block", fake_relevance),
|
("_levels_block", fake_levels), ("_relevance_block", fake_relevance),
|
||||||
("_question_pattern_block", fake_pattern), ("_artefacts_block", fake_artefacts),
|
("_question_pattern_block", fake_pattern), ("_artefacts_block", fake_artefacts),
|
||||||
("_outline_block", fake_outline)]:
|
("_outline_block", fake_outline), ("_konsolidiere_subblocks", fake_konsolidierung)]:
|
||||||
monkeypatch.setattr(ba, name, fn)
|
monkeypatch.setattr(ba, name, fn)
|
||||||
|
|
||||||
|
class _EmbOff: # Cross-Block-Barrier reicht ohne Modell alle Karten durch
|
||||||
|
@staticmethod
|
||||||
|
def available():
|
||||||
|
return False
|
||||||
|
monkeypatch.setattr(ba, "embedding", _EmbOff)
|
||||||
|
|
||||||
work = tmp_path / "arbeit"
|
work = tmp_path / "arbeit"
|
||||||
work.mkdir()
|
work.mkdir()
|
||||||
files = {"arbeit": work, "final": tmp_path / "blocks.md",
|
files = {"arbeit": work, "final": tmp_path / "blocks.md",
|
||||||
@@ -151,6 +170,10 @@ async def test_board1_full_flow(board_env):
|
|||||||
assert {s["sub_title"] for s in subs} == {"Sub Eins", "Sub Zwei"}
|
assert {s["sub_title"] for s in subs} == {"Sub Eins", "Sub Zwei"}
|
||||||
outline = await db.get_outline(TOPIC)
|
outline = await db.get_outline(TOPIC)
|
||||||
assert outline and "Kapitel 1" in outline
|
assert outline and "Kapitel 1" in outline
|
||||||
|
# Lauf-Summary am Flow-Ende: run_id + Zähler (QA diffed dagegen)
|
||||||
|
summary = json.loads((files["arbeit"] / "lauf-summary.json").read_text(encoding="utf-8"))
|
||||||
|
assert summary["run_id"] and summary["topic"] == TOPIC
|
||||||
|
assert summary["boards"].get("inventory", {}).get("done_block") == 4
|
||||||
|
|
||||||
|
|
||||||
async def test_filter_judges_run_parallel(board_env, monkeypatch):
|
async def test_filter_judges_run_parallel(board_env, monkeypatch):
|
||||||
@@ -839,3 +862,204 @@ async def test_ingest_strips_markdown_title(testdb, tmp_path):
|
|||||||
assert n == 1
|
assert n == 1
|
||||||
card = await testdb.kanban_get_card(TOPIC, B, "listscheduling")
|
card = await testdb.kanban_get_card(TOPIC, B, "listscheduling")
|
||||||
assert card["payload"]["title"] == "ListScheduling"
|
assert card["payload"]["title"] == "ListScheduling"
|
||||||
|
|
||||||
|
|
||||||
|
# ── QA-Gate: Inventar-Prüfung vor Board 2 ────────────────────────────────────────────
|
||||||
|
|
||||||
|
async def _run_flow(ctx, files, timeout=30, **kw):
|
||||||
|
import asyncio
|
||||||
|
return await asyncio.wait_for(
|
||||||
|
bi.run_boards(ctx, lambda *a, **k: None, files, {"type": "thema"}, None, "", research=False, **kw),
|
||||||
|
timeout=timeout)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_qa_gate_pauses_on_bad_note(board_env, monkeypatch):
|
||||||
|
"""Note unter Schwelle → Flow endet sauber, Board-2-Karten warten in subblocks."""
|
||||||
|
import qa as qa_mod
|
||||||
|
db, ctx, files = board_env
|
||||||
|
|
||||||
|
async def bad_qa(topic, llm=False):
|
||||||
|
return {"note": 5.0, "topic": TOPIC, "quoten": {}, "fremd": [], "artefakte": {}}
|
||||||
|
monkeypatch.setattr(qa_mod, "qa_report", bad_qa)
|
||||||
|
monkeypatch.setattr(qa_mod, "_write_report", lambda r: None)
|
||||||
|
monkeypatch.setattr(bi, "_QA_GATE_POLL", 0.05)
|
||||||
|
await _seed(db)
|
||||||
|
ok = await _run_flow(ctx, files)
|
||||||
|
assert ok
|
||||||
|
warten = await db.kanban_cards(TOPIC, board="artefacts", stage="subblocks")
|
||||||
|
assert len(warten) == 4 # alle Blöcke gespawnt, keiner verarbeitet
|
||||||
|
assert await db.kanban_count(TOPIC, "done_artefact", board="artefacts") == 0
|
||||||
|
|
||||||
|
|
||||||
|
async def test_qa_gate_force_overrides(board_env, monkeypatch):
|
||||||
|
"""qa_force=True („Trotzdem fortsetzen") übersteuert die schlechte Note."""
|
||||||
|
import qa as qa_mod
|
||||||
|
db, ctx, files = board_env
|
||||||
|
called = {"n": 0}
|
||||||
|
|
||||||
|
async def bad_qa(topic, llm=False):
|
||||||
|
called["n"] += 1
|
||||||
|
return {"note": 5.0, "topic": TOPIC, "quoten": {}, "fremd": [], "artefakte": {}}
|
||||||
|
monkeypatch.setattr(qa_mod, "qa_report", bad_qa)
|
||||||
|
monkeypatch.setattr(qa_mod, "_write_report", lambda r: None)
|
||||||
|
monkeypatch.setattr(bi, "_QA_GATE_POLL", 0.05)
|
||||||
|
await _seed(db)
|
||||||
|
ok = await _run_flow(ctx, files, qa_force=True)
|
||||||
|
assert ok
|
||||||
|
assert await db.kanban_count(TOPIC, "done_artefact", board="artefacts") >= 5
|
||||||
|
assert called["n"] <= 1 # Gate-Lauf übersprungen; höchstens Abschluss-QA
|
||||||
|
|
||||||
|
|
||||||
|
async def test_qa_gate_off_means_no_qa_call(board_env, monkeypatch):
|
||||||
|
import qa as qa_mod
|
||||||
|
db, ctx, files = board_env
|
||||||
|
called = {"n": 0}
|
||||||
|
|
||||||
|
async def spy_qa(topic, llm=False):
|
||||||
|
called["n"] += 1
|
||||||
|
return {"note": 10.0, "topic": TOPIC, "quoten": {}, "fremd": [], "artefakte": {}}
|
||||||
|
monkeypatch.setattr(qa_mod, "qa_report", spy_qa)
|
||||||
|
monkeypatch.setattr(qa_mod, "_write_report", lambda r: None)
|
||||||
|
monkeypatch.setattr(bi, "QA_GATE_NOTE", 0)
|
||||||
|
await _seed(db)
|
||||||
|
ok = await _run_flow(ctx, files)
|
||||||
|
assert ok
|
||||||
|
assert called["n"] == 1 # kein Gate-Lauf; nur die Abschluss-QA der Lauf-Summary
|
||||||
|
|
||||||
|
|
||||||
|
async def test_qa_gate_fail_open(board_env, monkeypatch):
|
||||||
|
"""QA crasht → Gate öffnet, Flow läuft komplett durch (fail-open)."""
|
||||||
|
import qa as qa_mod
|
||||||
|
db, ctx, files = board_env
|
||||||
|
|
||||||
|
async def broken_qa(topic, llm=False):
|
||||||
|
raise RuntimeError("kaputt")
|
||||||
|
monkeypatch.setattr(qa_mod, "qa_report", broken_qa)
|
||||||
|
monkeypatch.setattr(bi, "_QA_GATE_POLL", 0.05)
|
||||||
|
await _seed(db)
|
||||||
|
ok = await _run_flow(ctx, files)
|
||||||
|
assert ok
|
||||||
|
assert await db.kanban_count(TOPIC, "done_artefact", board="artefacts") >= 5
|
||||||
|
|
||||||
|
|
||||||
|
def test_qa_view_pausiert_logic(tmp_path, monkeypatch):
|
||||||
|
import qa as qa_mod
|
||||||
|
import json as _json
|
||||||
|
monkeypatch.setattr(qa_mod, "QA_DIR", tmp_path)
|
||||||
|
(tmp_path / TOPIC).mkdir()
|
||||||
|
(tmp_path / TOPIC / "r1.json").write_text(_json.dumps(
|
||||||
|
{"note": 5.0, "quoten": {"fremd": 0.2}, "fremd": ["X"], "unecht": ["Y"]}), encoding="utf-8")
|
||||||
|
counts = {"inventory": {"done_block": 3}, "artefacts": {"subblocks": 4}}
|
||||||
|
v = bi._qa_view(TOPIC, counts, None)
|
||||||
|
assert v["pausiert"] is True and v["note"] == 5.0 and v["befunde"] == ["X", "Y"]
|
||||||
|
from types import SimpleNamespace
|
||||||
|
laufend = SimpleNamespace(state={})
|
||||||
|
assert bi._qa_view(TOPIC, counts, laufend)["pausiert"] is False # Flow läuft noch
|
||||||
|
# Bausteine gelöscht → kein Badge, obwohl der Report noch existiert
|
||||||
|
assert bi._qa_view(TOPIC, {}, None) is None
|
||||||
|
|
||||||
|
|
||||||
|
def test_qa_view_picks_newest_by_mtime(tmp_path, monkeypatch):
|
||||||
|
"""Run-id-Namen (…-1311-5e5c) sortieren lexikographisch VOR Zeitstempel-Namen —
|
||||||
|
ein Re-Run überschreibt die run-id-Datei, das Badge muss trotzdem sie zeigen."""
|
||||||
|
import os
|
||||||
|
import qa as qa_mod
|
||||||
|
import json as _json
|
||||||
|
monkeypatch.setattr(qa_mod, "QA_DIR", tmp_path)
|
||||||
|
(tmp_path / TOPIC).mkdir()
|
||||||
|
alt = tmp_path / TOPIC / "20260703-141649.json"
|
||||||
|
alt.write_text(_json.dumps({"note": 10.0, "quoten": {}}), encoding="utf-8")
|
||||||
|
os.utime(alt, (1000, 1000))
|
||||||
|
neu = tmp_path / TOPIC / "20260703-1311-5e5c.json"
|
||||||
|
neu.write_text(_json.dumps({"note": 8.9, "quoten": {}}), encoding="utf-8")
|
||||||
|
os.utime(neu, (2000, 2000))
|
||||||
|
v = bi._qa_view(TOPIC, {"inventory": {"done_block": 3}}, None)
|
||||||
|
assert v["note"] == 8.9
|
||||||
|
|
||||||
|
|
||||||
|
async def test_supplement_material_mode_for_source_topics(board_env, tmp_path, monkeypatch):
|
||||||
|
"""Quellen-Thema: Supplement vergleicht gegen das MATERIAL (files, kein Web);
|
||||||
|
thema-Modus behält die Websuche (voller Zugriff)."""
|
||||||
|
db, ctx, files = board_env
|
||||||
|
seen = {}
|
||||||
|
|
||||||
|
async def spy_slot(ctx2, label, *, key, prompt, role, capabilities, payload, timeout, on_line=None):
|
||||||
|
seen[key] = (capabilities, prompt)
|
||||||
|
m = _PATH_RE.search(prompt)
|
||||||
|
if m and "-supplement" in key and "-beleg" not in key:
|
||||||
|
with open(m.group(1), "w", encoding="utf-8") as f:
|
||||||
|
json.dump({"blocks": []}, f)
|
||||||
|
return "ok", payload(None)
|
||||||
|
|
||||||
|
monkeypatch.setattr(bi, "run_single_slot", spy_slot)
|
||||||
|
flow = _mk_flow(tmp_path)
|
||||||
|
korpus = tmp_path / "korpus"
|
||||||
|
korpus.mkdir()
|
||||||
|
monkeypatch.setattr(bi, "source_folder", lambda t: korpus)
|
||||||
|
await bi._supplement_producer(ctx, flow, ["Alpha"])
|
||||||
|
caps, prompt = seen[f"blocks-{TOPIC}-supplement"]
|
||||||
|
assert caps == "files"
|
||||||
|
assert "LEARNING MATERIAL" in prompt and "Do NOT search the web" in prompt
|
||||||
|
|
||||||
|
seen.clear()
|
||||||
|
(tmp_path / "supplement.json").unlink() # Resume-Guard zurücksetzen
|
||||||
|
monkeypatch.setattr(bi, "source_folder", lambda t: None)
|
||||||
|
await bi._supplement_producer(ctx, flow, ["Alpha"])
|
||||||
|
caps, prompt = seen[f"blocks-{TOPIC}-supplement"]
|
||||||
|
assert caps == "full"
|
||||||
|
assert "Research the subject area" in prompt
|
||||||
|
|
||||||
|
|
||||||
|
# ── Anker-Gate: Quorum-Titel ohne Korpus-Beleg (Reader-Ko-Halluzination) ────────────
|
||||||
|
|
||||||
|
async def _anker_env(db, tmp_path, monkeypatch, titel_map):
|
||||||
|
(tmp_path / "korpus.txt").write_text("Der Graph ist zusammenhängend und endlich.", encoding="utf-8")
|
||||||
|
monkeypatch.setattr(bi, "source_folder", lambda t: tmp_path)
|
||||||
|
|
||||||
|
async def fake_members(topic, cid):
|
||||||
|
return [{"title": titel_map[cid], "description": "", "readers": ["r1", "r2"], "supplement": False}]
|
||||||
|
|
||||||
|
monkeypatch.setattr(bi, "_member_rows", fake_members)
|
||||||
|
monkeypatch.setattr(bi, "_rep", lambda rows: rows[0])
|
||||||
|
for cid in titel_map:
|
||||||
|
await db.kanban_upsert_card(TOPIC, B, cid, "cluster", "consensus_gate", {})
|
||||||
|
ctx = GenContext(topic=TOPIC, provider="claude", is_cancelled=lambda: False)
|
||||||
|
return ctx, [{"card_id": c, "payload": {}} for c in titel_map]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_anker_gate_rejects_unbelegtes(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Titel ohne Korpus-Anker → Beleg-Judge; „nein" → rejected/kein-beleg.
|
||||||
|
Titel MIT Anker geht ohne Judge nach naming."""
|
||||||
|
db = testdb
|
||||||
|
ctx, cards = await _anker_env(db, tmp_path, monkeypatch,
|
||||||
|
{"c1": "Graph Zusammenhang", "c2": "Königsberger Brückenproblem"})
|
||||||
|
|
||||||
|
async def fake_slot(ctx2, label, *, key, prompt, role, capabilities, payload, timeout):
|
||||||
|
assert "Brückenproblem" in prompt and "Zusammenhang" not in prompt # nur der Anker-lose
|
||||||
|
return "ok", payload((0, json.dumps({"relevant": {"1": "nein"}}), ""))
|
||||||
|
|
||||||
|
monkeypatch.setattr(bi, "run_single_slot", fake_slot)
|
||||||
|
await bi._proc_consensus_gate(ctx, _mk_flow(tmp_path), cards)
|
||||||
|
assert (await db.kanban_get_card(TOPIC, B, "c1"))["stage"] == "naming"
|
||||||
|
c2 = await db.kanban_get_card(TOPIC, B, "c2")
|
||||||
|
assert c2["stage"] == "rejected" and c2["payload"]["reason"] == "kein-beleg"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_anker_gate_fail_open(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Judge-Ausfall → Titel bleibt (2-Reader-Rückhalt)."""
|
||||||
|
db = testdb
|
||||||
|
ctx, cards = await _anker_env(db, tmp_path, monkeypatch, {"c9": "Königsberger Brückenproblem"})
|
||||||
|
|
||||||
|
async def broken_slot(*a, **kw):
|
||||||
|
return "failed", None
|
||||||
|
|
||||||
|
monkeypatch.setattr(bi, "run_single_slot", broken_slot)
|
||||||
|
await bi._proc_consensus_gate(ctx, _mk_flow(tmp_path), cards)
|
||||||
|
assert (await db.kanban_get_card(TOPIC, B, "c9"))["stage"] == "naming"
|
||||||
|
|
||||||
|
|
||||||
|
def test_hat_anker_ziffern_suffix():
|
||||||
|
ctoks = {"tsp", "graph", "kanten"}
|
||||||
|
assert bi._hat_anker("ΔTSP1-Algorithmus", ctoks) # tsp1 → tsp
|
||||||
|
assert not bi._hat_anker("Königsberger Brückenproblem", ctoks)
|
||||||
|
assert not bi._hat_anker("Algorithmus Verfahren", ctoks) # nur Stopwörter → kein Anker
|
||||||
|
|||||||
@@ -283,3 +283,37 @@ async def test_run_agent_logs_opencode_tokens(testdb, monkeypatch):
|
|||||||
rc, *_ = await agents.run_agent("blocks-t-tok", "p", 5, provider="minimax", scope=TOPIC)
|
rc, *_ = await agents.run_agent("blocks-t-tok", "p", 5, provider="minimax", scope=TOPIC)
|
||||||
assert rc == 0
|
assert rc == 0
|
||||||
assert recorded and recorded[0]["meta"]["tokens"]["cache_read"] == 100
|
assert recorded and recorded[0]["meta"]["tokens"]["cache_read"] == 100
|
||||||
|
|
||||||
|
|
||||||
|
# ── run_id-Registry + Lauf-Summary ───────────────────────────────────────────────────
|
||||||
|
|
||||||
|
async def test_run_id_stamped_on_events(testdb):
|
||||||
|
"""Registry gesetzt → Agent- und Stage-Events tragen die run_id; geleert → leer."""
|
||||||
|
db = testdb
|
||||||
|
db.set_current_run(TOPIC, "20260703-1200-abcd")
|
||||||
|
await db.add_event(TOPIC, "agent", key="k1", status="ok",
|
||||||
|
meta={"tokens": {"input": 10, "output": 2, "cache_read": 50, "cache_write": 1}})
|
||||||
|
await db.kanban_upsert_card(TOPIC, "inventory", "c1", "block", "ingest", {})
|
||||||
|
await db.kanban_advance(TOPIC, "inventory", "c1", "cluster")
|
||||||
|
db.set_current_run(TOPIC, None)
|
||||||
|
await db.add_event(TOPIC, "agent", key="k2", status="ok")
|
||||||
|
conn = await db.get_db()
|
||||||
|
rows = await (await conn.execute("SELECT key, run_id FROM events WHERE topic=? ORDER BY id", (TOPIC,))).fetchall()
|
||||||
|
by_key = {k: r for k, r in rows}
|
||||||
|
assert by_key["k1"] == "20260703-1200-abcd"
|
||||||
|
assert by_key["inventory:c1"] == "20260703-1200-abcd"
|
||||||
|
assert by_key["k2"] == ""
|
||||||
|
|
||||||
|
|
||||||
|
async def test_events_run_summary_aggregates(testdb):
|
||||||
|
db = testdb
|
||||||
|
db.set_current_run(TOPIC, "r1")
|
||||||
|
await db.add_event(TOPIC, "agent", key="a", status="ok", dur_ms=1000,
|
||||||
|
meta={"tokens": {"input": 10, "output": 2, "cache_read": 50, "cache_write": 1}})
|
||||||
|
await db.add_event(TOPIC, "agent", key="b", status="timeout", dur_ms=120000,
|
||||||
|
meta={"tokens": {"input": 5, "output": 0, "cache_read": 30, "cache_write": 0}})
|
||||||
|
db.set_current_run(TOPIC, None)
|
||||||
|
s = await db.events_run_summary(TOPIC, "r1")
|
||||||
|
assert s["agents"]["gesamt"] == 2 and s["agents"]["ok"] == 1 and s["agents"]["timeout"] == 1
|
||||||
|
assert s["agents"]["verlorene_min"] == 2
|
||||||
|
assert s["tokens"] == {"input": 15, "output": 2, "cache_read": 80, "cache_write": 1}
|
||||||
|
|||||||
@@ -150,8 +150,8 @@ def test_writer_template_has_examples_placeholder():
|
|||||||
from pipeline import _prompt
|
from pipeline import _prompt
|
||||||
text = _prompt("Guide-Writer-Board", topic="t", format_name="Guide", chapter="K1",
|
text = _prompt("Guide-Writer-Board", topic="t", format_name="Guide", chapter="K1",
|
||||||
assignment="- B", ziele="- z", facts="F", examples="", gaps="",
|
assignment="- B", ziele="- z", facts="F", examples="", gaps="",
|
||||||
spec="", out_path="/tmp/x.md", extra="")
|
budget=2000, spec="", out_path="/tmp/x.md", extra="")
|
||||||
assert "VERIFIED FACTS" in text
|
assert "VERIFIED FACTS" in text and "2000 characters" in text
|
||||||
|
|
||||||
|
|
||||||
async def test_fakten_gate_counts_examples_as_facts(testdb, monkeypatch, tmp_path):
|
async def test_fakten_gate_counts_examples_as_facts(testdb, monkeypatch, tmp_path):
|
||||||
@@ -222,3 +222,78 @@ async def test_writer_splits_oversized_first_draft(testdb, monkeypatch, tmp_path
|
|||||||
secs = _parse_fragment(card["md"])
|
secs = _parse_fragment(card["md"])
|
||||||
assert len(secs) == 1 and [s["title"] for s in secs[0]["subs"]] == ["Sub 1", "Sub 2"]
|
assert len(secs) == 1 and [s["title"] for s in secs[0]["subs"]] == ["Sub 1", "Sub 2"]
|
||||||
assert card["stage"] == "fakten_gate"
|
assert card["stage"] == "fakten_gate"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_lernziele_retry_bei_leerer_liste(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Leere Ziele-Liste → genau EIN Ersatz-Versuch (Key-Suffix -2); dessen Ziele landen in der DB."""
|
||||||
|
db = testdb
|
||||||
|
await db.upsert_guide_card(TOPIC, FMT, "alpha", "Alpha")
|
||||||
|
env = gb._Env(None, "g-r", TOPIC, FMT, "", tmp_path / "Guide.json",
|
||||||
|
{"Alpha": [{"title": "S1", "level": "beginner"}]}, {}, "(quelle)", "spec")
|
||||||
|
card = {"block_norm": "alpha", "block": "Alpha", "writer_rounds": 0}
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
async def fake_slot(ctx, label, *, key, prompt, role, capabilities, payload, timeout):
|
||||||
|
calls.append(key)
|
||||||
|
if len(calls) == 1:
|
||||||
|
return gb.OK, []
|
||||||
|
return gb.OK, [{"id": "z1", "text": "Ziel", "sub": "S1"}]
|
||||||
|
|
||||||
|
monkeypatch.setattr(gb, "run_single_slot", fake_slot)
|
||||||
|
assert await gb._stage_lernziele(env, card)
|
||||||
|
assert len(calls) == 2 and calls[1].endswith("-2")
|
||||||
|
assert [z["ziel_id"] for z in await db.list_lernziele(TOPIC, "alpha")] == ["z1"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_lernziele_zweimal_leer_laeuft_weiter(testdb, tmp_path, monkeypatch):
|
||||||
|
db = testdb
|
||||||
|
await db.upsert_guide_card(TOPIC, FMT, "alpha", "Alpha")
|
||||||
|
env = gb._Env(None, "g-r2", TOPIC, FMT, "", tmp_path / "Guide.json", {"Alpha": []}, {}, "(q)", "spec")
|
||||||
|
card = {"block_norm": "alpha", "block": "Alpha", "writer_rounds": 0}
|
||||||
|
|
||||||
|
async def leer(ctx, label, *, key, prompt, role, capabilities, payload, timeout):
|
||||||
|
return gb.OK, []
|
||||||
|
|
||||||
|
monkeypatch.setattr(gb, "run_single_slot", leer)
|
||||||
|
assert await gb._stage_lernziele(env, card)
|
||||||
|
assert (await db.list_guide_cards(TOPIC, FMT))[0]["stage"] == "zuweisung"
|
||||||
|
assert not await db.list_lernziele(TOPIC, "alpha")
|
||||||
|
|
||||||
|
|
||||||
|
async def test_lese_check_text_sink(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Lese-Check antwortet als Text, Engine-Sink persistiert; capabilities none."""
|
||||||
|
db = testdb
|
||||||
|
await db.upsert_guide_card(TOPIC, FMT, "alpha", "Alpha")
|
||||||
|
env = gb._Env(None, "g-l", TOPIC, FMT, "", tmp_path / "Guide.json", {"Alpha": []}, {}, "(q)", "spec")
|
||||||
|
md = ("<!-- kapitel: K -->\n<!-- section: Alpha -->\n<!-- compact -->\n- x\n"
|
||||||
|
"<!-- ausführlich -->\nText.")
|
||||||
|
card = {"block_norm": "alpha", "block": "Alpha", "writer_rounds": 0, "md": md}
|
||||||
|
seen = {}
|
||||||
|
|
||||||
|
async def fake_slot(ctx, label, *, key, prompt, role, capabilities, payload, timeout):
|
||||||
|
seen["caps"] = capabilities
|
||||||
|
return gb.OK, payload((0, '{"ok": true}', ""))
|
||||||
|
|
||||||
|
monkeypatch.setattr(gb, "run_single_slot", fake_slot)
|
||||||
|
monkeypatch.setattr(gb, "READABILITY_ACTIVE", False)
|
||||||
|
assert await gb._stage_lesbarkeit(env, card)
|
||||||
|
assert seen["caps"] == "none"
|
||||||
|
assert (await db.list_guide_cards(TOPIC, FMT))[0]["stage"] == "done"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_writer_prompt_traegt_budget(testdb, tmp_path, monkeypatch):
|
||||||
|
db = testdb
|
||||||
|
await db.upsert_guide_card(TOPIC, FMT, "alpha", "Alpha")
|
||||||
|
env = gb._Env(None, "g-w", TOPIC, FMT, "", tmp_path / "Guide.json",
|
||||||
|
{"Alpha": [{"title": "S1", "level": "beginner"},
|
||||||
|
{"title": "S2", "level": "beginner"}]}, {}, "(q)", "spec")
|
||||||
|
card = {"block_norm": "alpha", "block": "Alpha", "writer_rounds": 0, "chapter": "K", "gate_info": ""}
|
||||||
|
seen = {}
|
||||||
|
|
||||||
|
async def fake_slot(ctx, label, *, key, prompt, role, capabilities, payload, timeout):
|
||||||
|
seen["prompt"] = prompt
|
||||||
|
return gb.FAILED, None
|
||||||
|
|
||||||
|
monkeypatch.setattr(gb, "run_single_slot", fake_slot)
|
||||||
|
await gb._stage_writer(env, card)
|
||||||
|
assert str(gb._writer_budget(2)) in seen["prompt"] # 800 + 2×400
|
||||||
|
|||||||
65
backend/tests/test_guide_qa.py
Normal file
65
backend/tests/test_guide_qa.py
Normal file
@@ -0,0 +1,65 @@
|
|||||||
|
"""Guide-QA: Fehler-Injektion auf Mini-Guide-Karten — deterministisch, ohne LLM."""
|
||||||
|
|
||||||
|
import guide_qa as gq
|
||||||
|
import qa
|
||||||
|
|
||||||
|
|
||||||
|
def _card(block, md):
|
||||||
|
return {"block": block, "block_norm": block.casefold(), "md": md}
|
||||||
|
|
||||||
|
|
||||||
|
AUSF = ("<!-- section: Alpha -->\n<!-- compact -->\n- m\n<!-- ausführlich -->\n"
|
||||||
|
"Einstieg in den Block.\n"
|
||||||
|
"<!-- sub: beginner | Kantenzug Definition -->\n"
|
||||||
|
"Ein Kantenzug verbindet Knoten über Kanten im Graphen.\n")
|
||||||
|
|
||||||
|
|
||||||
|
def test_ausfuehrlich_extrahiert_lerntext():
|
||||||
|
assert gq._ausfuehrlich(AUSF).startswith("\nEinstieg")
|
||||||
|
assert gq._ausfuehrlich("nur text") == "nur text"
|
||||||
|
|
||||||
|
|
||||||
|
def test_marker_fehlend():
|
||||||
|
cards = [_card("Alpha", AUSF)]
|
||||||
|
rel = {"alpha": {"kantenzug definition", "fehlender aspekt"}}
|
||||||
|
out = gq.marker_fehlend(cards, rel)
|
||||||
|
assert out == ["Alpha · fehlender aspekt"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_ziel_ohne_anker():
|
||||||
|
cards = [_card("Alpha", AUSF)]
|
||||||
|
ziele = [{"block_norm": "alpha", "ziel_id": "z1", "text": "Kantenzug im Graphen erklären"},
|
||||||
|
{"block_norm": "alpha", "ziel_id": "z2", "text": "Adjazenzmatrix aufstellen können"}]
|
||||||
|
out = gq.ziel_ohne_anker(cards, ziele)
|
||||||
|
assert len(out) == 1 and "z2" in out[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_laengen_ausreisser():
|
||||||
|
duenn = _card("Alpha", "<!-- ausführlich -->\nkurz")
|
||||||
|
ok = _card("Beta", "<!-- ausführlich -->\n" + "x" * 500)
|
||||||
|
out = gq.laengen_ausreisser([duenn, ok], {"alpha": {"s1"}, "beta": {"s1"}})
|
||||||
|
assert [x["block"] for x in out] == ["Alpha"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_redundanz_findet_absatz_doppel():
|
||||||
|
a = "Der Kantenzug verbindet Knoten über mehrere Kanten und darf Knoten wiederholen. " * 3
|
||||||
|
b = "Der Kantenzug verbindet Knoten über mehrere Kanten und darf Knoten wiederholen, genau. " * 3
|
||||||
|
c = "Völlig anderes Thema: Matrizen, Determinanten und lineare Abbildungen im Vektorraum. " * 3
|
||||||
|
cards = [_card("Alpha", f"<!-- ausführlich -->\n{a}\n\n{c}"),
|
||||||
|
_card("Beta", f"<!-- ausführlich -->\n{b}")]
|
||||||
|
out = gq.redundanz(cards)
|
||||||
|
assert len(out) == 1 and out[0]["a"].startswith("Alpha")
|
||||||
|
|
||||||
|
|
||||||
|
def test_lesbarkeit_fail_open(monkeypatch):
|
||||||
|
def kaputt(md_by_num):
|
||||||
|
raise RuntimeError("Modell fehlt")
|
||||||
|
monkeypatch.setattr(gq.readability, "rate_sections", kaputt)
|
||||||
|
assert gq.lesbarkeit([_card("Alpha", AUSF)]) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_note_guide_kalibrierung():
|
||||||
|
"""Gewicht = Punktabzug bei 100 %: 10 % fachlich falsch × 3.0 → 7.0; ungemessen zählt nicht."""
|
||||||
|
assert qa.note({"fachlich_falsch": 0.1}, gq.NOTE_GEWICHTE_GUIDE) == 7.0
|
||||||
|
ohne = {"marker_fehlend": 0.0, "ziel_ohne_anker": 0.0}
|
||||||
|
assert qa.note(ohne, gq.NOTE_GEWICHTE_GUIDE) == 10.0
|
||||||
507
backend/tests/test_konsolidierung.py
Normal file
507
backend/tests/test_konsolidierung.py
Normal file
@@ -0,0 +1,507 @@
|
|||||||
|
"""Sub-Konsolidierung: In-Block-Panel (blocks._konsolidiere_subblocks) und
|
||||||
|
Cross-Block-Barrier (board_artefacts._proc_konsolidierung) — Judges gefaked, gegen Test-DB."""
|
||||||
|
|
||||||
|
import json
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
import blocks
|
||||||
|
import board_artefacts as ba
|
||||||
|
from kanban import Flow
|
||||||
|
from pipeline import FAILED, OK, GenContext
|
||||||
|
|
||||||
|
TOPIC = "konsolidierung"
|
||||||
|
|
||||||
|
|
||||||
|
def _ctx():
|
||||||
|
return GenContext(topic=TOPIC, provider="test", is_cancelled=lambda: False)
|
||||||
|
|
||||||
|
|
||||||
|
def _fake_slot(antworten):
|
||||||
|
"""run_single_slot-Fake: pro Judge-Key eine Antwort; schreibt via payload (wie der Engine-Sink)."""
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
async def fake(ctx, label, *, key, prompt, role, capabilities, payload, timeout):
|
||||||
|
calls.append({"key": key, "prompt": prompt})
|
||||||
|
j = key.rsplit("-", 1)[-1] # "j1"/"j2"
|
||||||
|
antwort = antworten.get(j)
|
||||||
|
if antwort is None:
|
||||||
|
return FAILED, None
|
||||||
|
return OK, payload((0, json.dumps(antwort), ""))
|
||||||
|
|
||||||
|
fake.calls = calls
|
||||||
|
return fake
|
||||||
|
|
||||||
|
|
||||||
|
async def _seed_block(db, bnorm, subs):
|
||||||
|
for s in subs:
|
||||||
|
await db.put_subblock(TOPIC, bnorm, blocks._norm_title(s), bnorm.title(), s, status="consensus")
|
||||||
|
|
||||||
|
|
||||||
|
# ── In-Block ────────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
async def test_merge_on_unanimity(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Beide Judges gruppieren 1+2 → Gewinner (mehr key_points) bleibt, facts-Union,
|
||||||
|
Verlierer wird DB-variant und fliegt aus raw/facts_map."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Durchstreichung: ~~text~~", "Durchstreichung: ~~text~~ streicht Text durch", "Fett: **text**"]
|
||||||
|
await _seed_block(db, "betonung", subs)
|
||||||
|
raw = {"Betonung": list(subs)}
|
||||||
|
facts = {"Betonung": {
|
||||||
|
blocks._norm_title(subs[0]): {"key_points": ["kp-a"], "cited_facts": [{"text": "z1"}]},
|
||||||
|
blocks._norm_title(subs[1]): {"key_points": ["kp-b", "kp-c"], "cited_facts": [{"text": "z1"}, {"text": "z2"}]},
|
||||||
|
}}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [[1, 2]], "luecken": ["Marker-Escaping fehlt"]},
|
||||||
|
"j2": {"gruppen": [[2, 1]], "luecken": ["Escaping von Markern"]}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, facts)
|
||||||
|
assert raw["Betonung"] == [subs[1], "Fett: **text**"] # Gewinner: 2 key_points > 1
|
||||||
|
wf = facts["Betonung"][blocks._norm_title(subs[1])]
|
||||||
|
assert wf["key_points"] == ["kp-b", "kp-c", "kp-a"]
|
||||||
|
assert wf["cited_facts"] == [{"text": "z1"}, {"text": "z2"}] # Union ohne Doppel
|
||||||
|
assert blocks._norm_title(subs[0]) not in facts["Betonung"]
|
||||||
|
rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, "betonung")}
|
||||||
|
assert rows[blocks._norm_title(subs[0])] == "variant"
|
||||||
|
assert rows[blocks._norm_title(subs[1])] == "consensus"
|
||||||
|
journale = list(tmp_path.glob("sub-konsolidierung-*.json"))
|
||||||
|
j = json.loads([p for p in journale if "-j" not in p.stem][0].read_text())
|
||||||
|
# Lücken-Schnitt: Token-Überlappung beider Judges, Formulierung von j1 gewinnt
|
||||||
|
assert j["gruppen"][0]["behalten"] == subs[1] and j["luecken"] == ["Marker-Escaping fehlt"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_dissent_keeps_everything(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Nur ein Judge gruppiert → keine Einstimmigkeit → kein Merge."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Eintrag eins", "Eintrag zwei"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [[1, 2]], "luecken": []},
|
||||||
|
"j2": {"gruppen": [], "luecken": []}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert raw["Block"] == subs
|
||||||
|
assert all(r["status"] == "consensus" for r in await db.list_subblocks(TOPIC, "block"))
|
||||||
|
|
||||||
|
|
||||||
|
async def test_judge_failure_fail_open(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Ein Judge UND der Ersatz ohne Ergebnis → fail-open, nichts ändert sich."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Eintrag eins", "Eintrag zwei"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [[1, 2]], "luecken": []}}) # j2 UND j3 → FAILED
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert raw["Block"] == subs
|
||||||
|
assert len(fake.calls) == 3 # j1, j2, Ersatz j3
|
||||||
|
|
||||||
|
|
||||||
|
async def test_ersatzrichter_bei_ausfall(testdb, tmp_path, monkeypatch):
|
||||||
|
"""j1 fällt aus → Ersatz j3 springt ein; Einstimmigkeit j2+j3 faltet.
|
||||||
|
Vorher entwertete EIN Timeout die gute Stimme (13 Links-Dubletten überlebten)."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Kurz", "Deutlich längerer Eintrag"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
fake = _fake_slot({"j2": {"gruppen": [[1, 2]], "luecken": []},
|
||||||
|
"j3": {"gruppen": [[2, 1]], "luecken": []}}) # j1 → FAILED
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert raw["Block"] == ["Deutlich längerer Eintrag"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_negation_guard_blocks_merge(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Gegensätzliche Aussagen werden selbst bei einstimmigen Judges nicht gefaltet."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Tabs werden expandiert", "Tabs werden nicht expandiert"]
|
||||||
|
await _seed_block(db, "tabs", subs)
|
||||||
|
raw = {"Tabs": list(subs)}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [[1, 2]], "luecken": []},
|
||||||
|
"j2": {"gruppen": [[1, 2]], "luecken": []}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert raw["Tabs"] == subs
|
||||||
|
|
||||||
|
|
||||||
|
async def test_resume_skips_judges(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Vorhandene j-Dateien → kein neuer Agenten-Call, Ergebnis wird übernommen."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Eintrag eins", "Eintrag zwei lang"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
import hashlib
|
||||||
|
h = hashlib.md5("\n".join(subs).encode()).hexdigest()[:8]
|
||||||
|
for j in (1, 2):
|
||||||
|
(tmp_path / f"sub-konsolidierung-{h}-j{j}.json").write_text(
|
||||||
|
json.dumps({"gruppen": [[1, 2]], "luecken": []}), encoding="utf-8")
|
||||||
|
|
||||||
|
async def kein_agent(*a, **kw):
|
||||||
|
raise AssertionError("Resume darf keinen Agenten starten")
|
||||||
|
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", kein_agent)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert raw["Block"] == ["Eintrag zwei lang"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_schema_accepts_both_group_forms():
|
||||||
|
"""Alte Listenform [1,4] und neue {haupt, weitere}-Form parsen beide; kataloge/fremd optional."""
|
||||||
|
alt = blocks._konsolidierung_schema({"gruppen": [[1, 4]], "luecken": []}, 5)
|
||||||
|
assert alt["gruppen"] == [{"haupt": None, "ids": [1, 4]}] and alt["fremd"] == set()
|
||||||
|
neu = blocks._konsolidierung_schema(
|
||||||
|
{"gruppen": [{"haupt": 4, "weitere": [1]}],
|
||||||
|
"kataloge": [{"titel": "Katalog: Symbole", "mitglieder": [2, 3]}],
|
||||||
|
"fremd": [5], "luecken": ["x"]}, 5)
|
||||||
|
assert neu["gruppen"] == [{"haupt": 4, "ids": [1, 4]}]
|
||||||
|
assert neu["kataloge"] == [{"titel": "Katalog: Symbole", "ids": [2, 3]}]
|
||||||
|
assert neu["fremd"] == {5} and neu["luecken"] == ["x"]
|
||||||
|
assert blocks._konsolidierung_schema({"gruppen": [{"haupt": 9, "weitere": [1]}]}, 5) == \
|
||||||
|
{"gruppen": [], "kataloge": [], "fremd": set(), "luecken": []} # id out of range
|
||||||
|
|
||||||
|
|
||||||
|
async def test_haupt_beats_heuristic(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Judges nennen den kürzeren Eintrag als haupt → er gewinnt trotz weniger key_points."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Basis", "Detailregel mit sehr langem Titel und Facts"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
facts = {"Block": {blocks._norm_title(subs[1]): {"key_points": ["a", "b", "c"]}}}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [{"haupt": 1, "weitere": [2]}], "luecken": []},
|
||||||
|
"j2": {"gruppen": [{"haupt": 1, "weitere": [2]}], "luecken": []}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, facts)
|
||||||
|
assert raw["Block"] == ["Basis"]
|
||||||
|
assert facts["Block"][blocks._norm_title("Basis")]["key_points"] == ["a", "b", "c"] # Union geerbt
|
||||||
|
|
||||||
|
|
||||||
|
async def test_katalog_bundles_to_new_row(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Einstimmige Katalog-Mitglieder → neue consensus-Zeile mit Facts-Union, Mitglieder variant."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Pfeilsymbole: a b c", "Mengensymbole: d e f", "Eigene Regel"]
|
||||||
|
await _seed_block(db, "mathe", subs)
|
||||||
|
raw = {"Mathe": list(subs)}
|
||||||
|
facts = {"Mathe": {blocks._norm_title(subs[0]): {"key_points": ["kp1"]},
|
||||||
|
blocks._norm_title(subs[1]): {"key_points": ["kp2"]}}}
|
||||||
|
kat = {"titel": "Symbolkatalog: Pfeile und Mengen", "mitglieder": [1, 2]}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [], "kataloge": [kat], "luecken": []},
|
||||||
|
"j2": {"gruppen": [], "kataloge": [kat], "luecken": []}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, facts)
|
||||||
|
assert raw["Mathe"] == ["Eigene Regel", "Symbolkatalog: Pfeile und Mengen"]
|
||||||
|
kn = blocks._norm_title("Symbolkatalog: Pfeile und Mengen")
|
||||||
|
assert sorted(facts["Mathe"][kn]["key_points"]) == ["kp1", "kp2"]
|
||||||
|
rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, "mathe")}
|
||||||
|
assert rows[kn] == "consensus"
|
||||||
|
assert rows[blocks._norm_title(subs[0])] == "variant"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_katalog_dissent_keeps_members(testdb, tmp_path, monkeypatch):
|
||||||
|
db = testdb
|
||||||
|
subs = ["Pfeilsymbole: a b c", "Mengensymbole: d e f"]
|
||||||
|
await _seed_block(db, "mathe", subs)
|
||||||
|
raw = {"Mathe": list(subs)}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [], "kataloge": [{"titel": "K", "mitglieder": [1, 2]}], "luecken": []},
|
||||||
|
"j2": {"gruppen": [], "kataloge": [], "luecken": []}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert raw["Mathe"] == subs
|
||||||
|
|
||||||
|
|
||||||
|
async def test_fremd_unanimous_discards(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Einstimmig fremd → discarded + raus; einseitig fremd → bleibt."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["CSS display überschreibt Verhalten", "Echte Markdown-Regel", "Nur einer hält es für fremd"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [], "fremd": [1, 3], "luecken": []},
|
||||||
|
"j2": {"gruppen": [], "fremd": [1], "luecken": []}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
luecken = await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert raw["Block"] == [subs[1], subs[2]]
|
||||||
|
rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, "block")}
|
||||||
|
assert rows[blocks._norm_title(subs[0])] == "discarded"
|
||||||
|
assert rows[blocks._norm_title(subs[2])] == "consensus"
|
||||||
|
assert luecken == {}
|
||||||
|
|
||||||
|
|
||||||
|
async def test_luecken_nur_bei_einstimmigkeit(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Nur Lücken mit Token-Überlappung BEIDER Judges überleben; einseitige fallen weg."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Eintrag eins", "Eintrag zwei"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [], "luecken": ["Inline-HTML fehlt", "Front-Matter"]},
|
||||||
|
"j2": {"gruppen": [], "luecken": ["nichts zu Inline-HTML"]}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
luecken = await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, {})
|
||||||
|
assert luecken == {"Block": ["Inline-HTML fehlt"]}
|
||||||
|
|
||||||
|
|
||||||
|
async def test_kp_deckel_im_judge_prompt(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Prompt zeigt max. 3 key_points je Sub (Timeout-Schutz); die Union bleibt voll."""
|
||||||
|
db = testdb
|
||||||
|
subs = ["Eintrag eins", "Eintrag zwei"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
facts = {"Block": {blocks._norm_title(subs[0]): {"key_points": [f"kp{i}" for i in range(1, 6)]}}}
|
||||||
|
fake = _fake_slot({"j1": {"gruppen": [], "luecken": []}, "j2": {"gruppen": [], "luecken": []}})
|
||||||
|
monkeypatch.setattr(blocks, "run_single_slot", fake)
|
||||||
|
await blocks._konsolidiere_subblocks(_ctx(), {"arbeit": tmp_path}, raw, facts)
|
||||||
|
prompt = fake.calls[0]["prompt"]
|
||||||
|
assert "kp3" in prompt and "kp4" not in prompt
|
||||||
|
|
||||||
|
|
||||||
|
def test_luecken_schnitt_cap():
|
||||||
|
l1 = [f"Aspekt-{k} fehlt" for k in ("eins", "zwei", "drei", "vier", "fünf")]
|
||||||
|
assert blocks._luecken_schnitt(l1, list(l1)) == l1[:3] # Cap 3
|
||||||
|
assert blocks._luecken_schnitt(["Inline-HTML"], ["Tabellen-Syntax"]) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_neg_set_lemmatisiert():
|
||||||
|
"""kein/keine/keinen falten auf einen Stamm; nicht vs. ohne bleiben verschieden."""
|
||||||
|
a = blocks._neg_set("Fehlerverhalten (kein Syntaxfehler)")
|
||||||
|
b = blocks._neg_set("Fehlerverhalten (keine Syntax-Fehlermeldung)")
|
||||||
|
assert a == b == frozenset({"kein"})
|
||||||
|
assert blocks._neg_set("nicht expandiert") != blocks._neg_set("ohne Expansion")
|
||||||
|
assert blocks._neg_set("niemals gerendert") == blocks._neg_set("nie gerendert")
|
||||||
|
|
||||||
|
|
||||||
|
# ── Lücken-Nachfass ─────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
async def _nachfass_env(db, monkeypatch, facts_result):
|
||||||
|
subs = ["Eintrag eins"]
|
||||||
|
await _seed_block(db, "block", subs)
|
||||||
|
raw = {"Block": list(subs)}
|
||||||
|
facts_map = {"Block": {}}
|
||||||
|
|
||||||
|
async def fake_race(topic, label, slots, quorum, timeout, provider, cancelled=None, grace=0):
|
||||||
|
return [{"Block": ["Eintrag eins", "Neuer Aspekt"]}]
|
||||||
|
|
||||||
|
async def fake_facts(ctx, set_p, files, fraw, q, folder, instructions, ns="", lbl="", sources=None, slim=False):
|
||||||
|
assert slim is True # Nachfass nutzt die schlanke Facts-Variante
|
||||||
|
assert list(fraw["Block"]) == ["Neuer Aspekt"] # nur der frische Fund geht ins Gate
|
||||||
|
return facts_result
|
||||||
|
|
||||||
|
monkeypatch.setattr(blocks, "_race", fake_race)
|
||||||
|
monkeypatch.setattr(blocks, "_facts_block", fake_facts)
|
||||||
|
monkeypatch.setattr(blocks, "EMBEDDING_AKTIV", False)
|
||||||
|
return raw, facts_map
|
||||||
|
|
||||||
|
|
||||||
|
async def test_nachfass_adopts_backed_find(testdb, tmp_path, monkeypatch):
|
||||||
|
db = testdb
|
||||||
|
nn = blocks._norm_title("Neuer Aspekt")
|
||||||
|
raw, facts_map = await _nachfass_env(db, monkeypatch,
|
||||||
|
({"Block": {nn: {"key_points": ["kp"]}}}, {}))
|
||||||
|
n = await blocks._luecken_runde(_ctx(), {"arbeit": tmp_path}, "Block", ["Aspekt"],
|
||||||
|
raw, facts_map, {"type": "thema"}, None)
|
||||||
|
assert n == 1 and raw["Block"] == ["Eintrag eins", "Neuer Aspekt"]
|
||||||
|
assert facts_map["Block"][nn]["key_points"] == ["kp"]
|
||||||
|
rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, "block")}
|
||||||
|
assert rows[nn] == "consensus"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_nachfass_drops_unbacked_find(testdb, tmp_path, monkeypatch):
|
||||||
|
db = testdb
|
||||||
|
nn = blocks._norm_title("Neuer Aspekt")
|
||||||
|
raw, facts_map = await _nachfass_env(db, monkeypatch, ({}, {"Block": {nn}}))
|
||||||
|
n = await blocks._luecken_runde(_ctx(), {"arbeit": tmp_path}, "Block", ["Aspekt"],
|
||||||
|
raw, facts_map, {"type": "thema"}, None)
|
||||||
|
assert n == 0 and raw["Block"] == ["Eintrag eins"]
|
||||||
|
assert not any(r["sub_norm"] == nn for r in await db.list_subblocks(TOPIC, "block"))
|
||||||
|
|
||||||
|
|
||||||
|
async def test_nachfass_drops_find_without_facts(testdb, tmp_path, monkeypatch):
|
||||||
|
"""HARTES Gate: kein Facts-Eintrag = kein Beleg = keine Übernahme — nicht nur
|
||||||
|
aktiv Verworfenes fliegt (Bilder-Lauf: 13 von 18 kamen ohne Beleg durch)."""
|
||||||
|
db = testdb
|
||||||
|
raw, facts_map = await _nachfass_env(db, monkeypatch, ({}, {})) # Facts fand NICHTS
|
||||||
|
n = await blocks._luecken_runde(_ctx(), {"arbeit": tmp_path}, "Block", ["Aspekt"],
|
||||||
|
raw, facts_map, {"type": "thema"}, None)
|
||||||
|
assert n == 0 and raw["Block"] == ["Eintrag eins"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_facts_stage_konsolidiert_nachfass_funde_erneut(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Kreis geschlossen: nach Übernahmen läuft die Konsolidierung ein zweites Mal;
|
||||||
|
deren Lücken lösen KEINEN weiteren Nachfass aus."""
|
||||||
|
db = testdb
|
||||||
|
payload = {"title": "Alpha", "raw": {"Alpha": ["s1"]}}
|
||||||
|
await db.kanban_upsert_card(TOPIC, "artefacts", "alpha", "ablock", "facts", payload)
|
||||||
|
calls = {"kons": 0, "nf": 0}
|
||||||
|
|
||||||
|
async def fake_facts(ctx, set_p, files, raw, q, folder, instructions, ns="", lbl="", sources=None, slim=False):
|
||||||
|
return {"Alpha": {}}, {}
|
||||||
|
|
||||||
|
async def fake_kons(ctx, files, raw, facts_map, instructions="", ns="", lbl=""):
|
||||||
|
calls["kons"] += 1
|
||||||
|
return {"Alpha": ["Lücke X"]} # meldet auch in Runde 2 — darf nicht erneut nachfassen
|
||||||
|
|
||||||
|
async def fake_nf(ctx, files, title, luecken, raw, facts_map, q, folder,
|
||||||
|
instructions="", ns="", lbl="", sources=None):
|
||||||
|
calls["nf"] += 1
|
||||||
|
return 2
|
||||||
|
|
||||||
|
monkeypatch.setattr(ba, "_facts_block", fake_facts)
|
||||||
|
monkeypatch.setattr(ba, "_konsolidiere_subblocks", fake_kons)
|
||||||
|
monkeypatch.setattr(ba, "_luecken_runde", fake_nf)
|
||||||
|
flow = Flow(TOPIC, work_dir=tmp_path)
|
||||||
|
await ba._proc_facts(_ctx(), flow, {"arbeit": tmp_path}, {"type": "thema"}, None, "",
|
||||||
|
[{"card_id": "alpha", "payload": payload}])
|
||||||
|
assert calls == {"kons": 2, "nf": 1}
|
||||||
|
assert (await db.kanban_get_card(TOPIC, "artefacts", "alpha"))["stage"] == "levels"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_finalize_purges_stale_rows(testdb, tmp_path):
|
||||||
|
"""Re-Run-Waisen: Finalize löscht Alt-Fragen/-Artefakte des Blocks vor dem Upsert."""
|
||||||
|
db = testdb
|
||||||
|
await db.upsert_question_pattern(TOPIC, "alpha", "alt-sub", "Alpha", "Alt", "Alte Frage?")
|
||||||
|
await db.put_sub_artifact(TOPIC, "alpha", "alt-sub", "flashcard", "{}", "Alpha", "Alt")
|
||||||
|
await db.kanban_upsert_card(TOPIC, "artefacts", "alpha", "ablock", "finalize", {})
|
||||||
|
files = {k: tmp_path / f"{k}.json" for k in
|
||||||
|
("sub_roh", "facts", "sidecar", "question_pattern", "artefakte")}
|
||||||
|
card = {"card_id": "alpha", "payload": {
|
||||||
|
"title": "Alpha", "raw": {"Alpha": ["Neu"]}, "facts": {},
|
||||||
|
"sidecar": {"Alpha": [{"title": "Neu", "level": "beginner"}]},
|
||||||
|
"pattern": {"Alpha": [{"subblock": "Neu", "question": "F?"}]},
|
||||||
|
"artefacts": {"flashcard": [{"block": "Alpha", "subblock": "Neu", "front": "F", "back": "B"}]}}}
|
||||||
|
flow = Flow(TOPIC, work_dir=tmp_path)
|
||||||
|
await ba._proc_finalize(_ctx(), flow, files, [card])
|
||||||
|
assert {r["sub_norm"] for r in await db.list_question_pattern(TOPIC)} == {"neu"}
|
||||||
|
assert {(r["sub_norm"], r["type"]) for r in await db.get_sub_artefakte(TOPIC)} == {("neu", "flashcard")}
|
||||||
|
|
||||||
|
|
||||||
|
# ── Cross-Block ─────────────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
class _FakeEmb:
|
||||||
|
"""Gleicher Text → gleicher Einheitsvektor, sonst orthogonal (cos 1.0 / 0.0)."""
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def available():
|
||||||
|
return True
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def embed_sims(texts):
|
||||||
|
uniq = {t: k for k, t in enumerate(dict.fromkeys(texts))}
|
||||||
|
arr = np.zeros((len(texts), max(len(uniq), 1)))
|
||||||
|
for r, t in enumerate(texts):
|
||||||
|
arr[r, uniq[t]] = 1.0
|
||||||
|
return arr @ arr.T
|
||||||
|
|
||||||
|
|
||||||
|
async def _cross_env(db, tmp_path):
|
||||||
|
flow = Flow(TOPIC, work_dir=tmp_path)
|
||||||
|
cards = []
|
||||||
|
for bnorm, subs in (("alpha", ["Gleiche Aussage", "Nur in Alpha"]),
|
||||||
|
("beta", ["Gleiche Aussage", "Nur in Beta"])):
|
||||||
|
payload = {"title": bnorm.title(),
|
||||||
|
"raw": {bnorm.title(): list(subs)},
|
||||||
|
"sidecar": {bnorm.title(): [{"title": s, "level": "beginner"} for s in subs]},
|
||||||
|
"facts": {bnorm.title(): {blocks._norm_title(s): {"key_points": [f"kp {s}"]} for s in subs}}}
|
||||||
|
await db.kanban_upsert_card(TOPIC, "artefacts", bnorm, "ablock", "konsolidierung", payload)
|
||||||
|
await _seed_block(db, bnorm, subs)
|
||||||
|
cards.append({"card_id": bnorm, "payload": payload})
|
||||||
|
return flow, cards
|
||||||
|
|
||||||
|
|
||||||
|
async def test_crossblock_folds_loser(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Einstimmig „a" → Beta verliert die geteilte Aussage, Karten wandern zu levels."""
|
||||||
|
db = testdb
|
||||||
|
flow, cards = await _cross_env(db, tmp_path)
|
||||||
|
monkeypatch.setattr(ba, "embedding", _FakeEmb)
|
||||||
|
fake = _fake_slot({"j1": {"pairs": {"1": "a"}}, "j2": {"pairs": {"1": "a"}}})
|
||||||
|
monkeypatch.setattr(ba, "run_single_slot", fake)
|
||||||
|
await ba._proc_konsolidierung(_ctx(), flow, {}, "", cards)
|
||||||
|
assert "Gleiche Aussage" in fake.calls[0]["prompt"]
|
||||||
|
beta = await db.kanban_get_card(TOPIC, "artefacts", "beta")
|
||||||
|
assert beta["stage"] == "question_pattern"
|
||||||
|
assert beta["payload"]["raw"]["Beta"] == ["Nur in Beta"]
|
||||||
|
assert blocks._norm_title("Gleiche Aussage") not in beta["payload"]["facts"]["Beta"]
|
||||||
|
# Barriere liegt jetzt hinter levels/relevance → auch die sidecar muss den Fold tragen
|
||||||
|
assert [e["title"] for e in beta["payload"]["sidecar"]["Beta"]] == ["Nur in Beta"]
|
||||||
|
alpha = await db.kanban_get_card(TOPIC, "artefacts", "alpha")
|
||||||
|
assert alpha["stage"] == "question_pattern"
|
||||||
|
assert alpha["payload"]["raw"]["Alpha"] == ["Gleiche Aussage", "Nur in Alpha"]
|
||||||
|
rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, "beta")}
|
||||||
|
assert rows[blocks._norm_title("Gleiche Aussage")] == "variant"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_crossblock_tiebreaker_folds(testdb, tmp_path, monkeypatch):
|
||||||
|
"""j1/j2 uneinig → j3 entscheidet mit Mehrheit; hier „a" → Beta verliert."""
|
||||||
|
db = testdb
|
||||||
|
flow, cards = await _cross_env(db, tmp_path)
|
||||||
|
monkeypatch.setattr(ba, "embedding", _FakeEmb)
|
||||||
|
fake = _fake_slot({"j1": {"pairs": {"1": "a"}}, "j2": {"pairs": {"1": "nein"}},
|
||||||
|
"j3": {"pairs": {"1": "a"}}})
|
||||||
|
monkeypatch.setattr(ba, "run_single_slot", fake)
|
||||||
|
await ba._proc_konsolidierung(_ctx(), flow, {}, "", cards)
|
||||||
|
assert len(fake.calls) == 3
|
||||||
|
beta = await db.kanban_get_card(TOPIC, "artefacts", "beta")
|
||||||
|
assert beta["payload"]["raw"]["Beta"] == ["Nur in Beta"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_crossblock_dissent_without_tiebreaker_keeps_both(testdb, tmp_path, monkeypatch):
|
||||||
|
"""j3 liefert nichts (FAILED) → fail-open, Paar bleibt."""
|
||||||
|
db = testdb
|
||||||
|
flow, cards = await _cross_env(db, tmp_path)
|
||||||
|
monkeypatch.setattr(ba, "embedding", _FakeEmb)
|
||||||
|
fake = _fake_slot({"j1": {"pairs": {"1": "a"}}, "j2": {"pairs": {"1": "b"}}}) # j3 fehlt → FAILED
|
||||||
|
monkeypatch.setattr(ba, "run_single_slot", fake)
|
||||||
|
await ba._proc_konsolidierung(_ctx(), flow, {}, "", cards)
|
||||||
|
beta = await db.kanban_get_card(TOPIC, "artefacts", "beta")
|
||||||
|
assert beta["stage"] == "question_pattern"
|
||||||
|
assert beta["payload"]["raw"]["Beta"] == ["Gleiche Aussage", "Nur in Beta"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_crossblock_ersatzrichter(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Nur ein Richter liefert → Ersatz jE als zweite Stimme; Einstimmigkeit faltet."""
|
||||||
|
db = testdb
|
||||||
|
flow, cards = await _cross_env(db, tmp_path)
|
||||||
|
monkeypatch.setattr(ba, "embedding", _FakeEmb)
|
||||||
|
fake = _fake_slot({"j1": {"pairs": {"1": "a"}}, "jE": {"pairs": {"1": "a"}}}) # j2 → FAILED
|
||||||
|
monkeypatch.setattr(ba, "run_single_slot", fake)
|
||||||
|
await ba._proc_konsolidierung(_ctx(), flow, {}, "", cards)
|
||||||
|
beta = await db.kanban_get_card(TOPIC, "artefacts", "beta")
|
||||||
|
assert beta["payload"]["raw"]["Beta"] == ["Nur in Beta"]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_crossblock_without_embedding_advances(testdb, tmp_path, monkeypatch):
|
||||||
|
db = testdb
|
||||||
|
flow, cards = await _cross_env(db, tmp_path)
|
||||||
|
|
||||||
|
class _Aus:
|
||||||
|
@staticmethod
|
||||||
|
def available():
|
||||||
|
return False
|
||||||
|
|
||||||
|
async def kein_agent(*a, **kw):
|
||||||
|
raise AssertionError("ohne Embedding kein Judge")
|
||||||
|
|
||||||
|
monkeypatch.setattr(ba, "embedding", _Aus)
|
||||||
|
monkeypatch.setattr(ba, "run_single_slot", kein_agent)
|
||||||
|
await ba._proc_konsolidierung(_ctx(), flow, {}, "", cards)
|
||||||
|
for cid in ("alpha", "beta"):
|
||||||
|
assert (await db.kanban_get_card(TOPIC, "artefacts", cid))["stage"] == "question_pattern"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_crossblock_context_wins(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Kontext-Sub (Block schon hinter der Barrier) gewinnt auch bei Verdict „b" —
|
||||||
|
die Paket-Seite fällt, der Kontext bleibt unangetastet."""
|
||||||
|
db = testdb
|
||||||
|
flow = Flow(TOPIC, work_dir=tmp_path)
|
||||||
|
payload = {"title": "Alpha", "raw": {"Alpha": ["Gleiche Aussage"]}, "facts": {"Alpha": {}}}
|
||||||
|
await db.kanban_upsert_card(TOPIC, "artefacts", "alpha", "ablock", "konsolidierung", payload)
|
||||||
|
await _seed_block(db, "alpha", ["Gleiche Aussage"])
|
||||||
|
cards = [{"card_id": "alpha", "payload": payload}]
|
||||||
|
# Kontext-Block "gamma" ist bereits weiter (Stage levels) und hält dieselbe Aussage
|
||||||
|
await db.kanban_upsert_card(TOPIC, "artefacts", "gamma", "ablock", "levels",
|
||||||
|
{"title": "Gamma", "raw": {"Gamma": ["Gleiche Aussage"]}, "facts": {}})
|
||||||
|
await _seed_block(db, "gamma", ["Gleiche Aussage"])
|
||||||
|
monkeypatch.setattr(ba, "embedding", _FakeEmb)
|
||||||
|
# Verdict „a": das Paket (A) soll behalten — Kontext faltet trotzdem nie
|
||||||
|
fake = _fake_slot({"j1": {"pairs": {"1": "a"}}, "j2": {"pairs": {"1": "a"}}})
|
||||||
|
monkeypatch.setattr(ba, "run_single_slot", fake)
|
||||||
|
await ba._proc_konsolidierung(_ctx(), flow, {}, "", cards)
|
||||||
|
alpha = await db.kanban_get_card(TOPIC, "artefacts", "alpha")
|
||||||
|
assert alpha["payload"]["raw"].get("Alpha", []) == [] # Paket-Seite gefaltet
|
||||||
|
gamma_rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, "gamma")}
|
||||||
|
assert gamma_rows[blocks._norm_title("Gleiche Aussage")] == "consensus" # Kontext unberührt
|
||||||
216
backend/tests/test_qa.py
Normal file
216
backend/tests/test_qa.py
Normal file
@@ -0,0 +1,216 @@
|
|||||||
|
"""QA-Detektoren: Fehler-Injektion auf Mini-Korpus — deterministisch, ohne LLM/Embedding."""
|
||||||
|
|
||||||
|
import qa
|
||||||
|
|
||||||
|
|
||||||
|
CORPUS = {"Skript.txt": (
|
||||||
|
"Kapitel 1: Vertex Cover — Definition, Approximation und Beweis der Guete.\n\n"
|
||||||
|
"Kapitel 2: Matching in Graphen — perfektes Matching und Augmentationswege.")}
|
||||||
|
BLOCKS = [
|
||||||
|
{"title": "Vertex Cover", "description": "Knotenüberdeckung", "sources": ["Skript.txt"]},
|
||||||
|
{"title": "Matching", "description": "Paarung in Graphen", "sources": ["Skript.txt"]},
|
||||||
|
]
|
||||||
|
SUBS = {"vertex cover": ["Approximation der Guete"], "matching": ["Matching in Graphen", "Augmentationswege"]}
|
||||||
|
|
||||||
|
|
||||||
|
def test_baseline_clean(monkeypatch):
|
||||||
|
"""Sauberes Soll-Inventar → alle Detektoren still (jeder Absatz ein Abschnitt)."""
|
||||||
|
monkeypatch.setattr(qa, "SECTION_CHARS", 20)
|
||||||
|
assert qa.dubletten(BLOCKS, emb_on=False) == []
|
||||||
|
assert qa.luecken(BLOCKS, SUBS, CORPUS) == []
|
||||||
|
assert qa.fremd(BLOCKS, CORPUS) == []
|
||||||
|
assert qa.hygiene(BLOCKS) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_injected_duplicate_found():
|
||||||
|
b = BLOCKS + [{"title": "Vertex-Cover-Problem", "description": "", "sources": []}]
|
||||||
|
pairs = qa.dubletten(b, emb_on=False)
|
||||||
|
assert any({p["a"], p["b"]} == {"Vertex Cover", "Vertex-Cover-Problem"} for p in pairs)
|
||||||
|
|
||||||
|
|
||||||
|
def test_acronym_signal():
|
||||||
|
b = BLOCKS + [{"title": "VC (Vertex Cover)", "description": "", "sources": []}]
|
||||||
|
pairs = qa.dubletten(b, emb_on=False)
|
||||||
|
hit = next(p for p in pairs if "VC (Vertex Cover)" in (p["a"], p["b"]) and "Vertex Cover" in (p["a"], p["b"]))
|
||||||
|
assert hit["signale"].get("akronym") is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_relation_operand_not_suspicious():
|
||||||
|
"""Relation vs. Operand ist per Design getrennt — kein Verdachtspaar."""
|
||||||
|
b = BLOCKS + [{"title": "3-SAT ≤ Vertex Cover", "description": "", "sources": []}]
|
||||||
|
pairs = qa.dubletten(b, emb_on=False)
|
||||||
|
assert not any("≤" in p["a"] + p["b"] for p in pairs)
|
||||||
|
|
||||||
|
|
||||||
|
def test_removed_block_creates_gap(monkeypatch):
|
||||||
|
monkeypatch.setattr(qa, "SECTION_CHARS", 20)
|
||||||
|
only_vc = [BLOCKS[0]]
|
||||||
|
gaps = qa.luecken(only_vc, {"vertex cover": SUBS["vertex cover"]}, CORPUS)
|
||||||
|
assert len(gaps) == 1 and "Matching" in gaps[0]["vorschau"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_foreign_block_flagged():
|
||||||
|
b = BLOCKS + [{"title": "Quantencomputer Grundlagen", "description": "", "sources": []}]
|
||||||
|
assert qa.fremd(b, CORPUS) == ["Quantencomputer Grundlagen"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_beleg_flags_unbacked_sub():
|
||||||
|
rows = [{"block": "Matching", "sub_title": "Erfunden", "mentions": 0, "status": "consensus"},
|
||||||
|
{"block": "Matching", "sub_title": "Belegt", "mentions": 3, "status": "consensus"}]
|
||||||
|
r = qa.beleg([{"title": "Matching", "sources": []}], rows)
|
||||||
|
assert r["subs_ohne_beleg"] == ["Matching · Erfunden"]
|
||||||
|
assert r["bloecke_ohne_quelle"] == ["Matching"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_hygiene_flags():
|
||||||
|
b = [{"title": "**Fett**", "description": "", "sources": []},
|
||||||
|
{"title": "Block (2)", "description": "ok", "sources": []}]
|
||||||
|
h = {x["titel"]: x["probleme"] for x in qa.hygiene(b)}
|
||||||
|
assert "markdown" in h["**Fett**"] and "leere-beschreibung" in h["**Fett**"]
|
||||||
|
assert h["Block (2)"] == ["kollisions-suffix"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_sections_split_on_paragraphs():
|
||||||
|
secs = qa._sections("a\n\nb\n\nc", goal=3)
|
||||||
|
assert len(secs) >= 2 and "".join(secs).replace("\n", "") == "abc"
|
||||||
|
|
||||||
|
|
||||||
|
def test_note_deterministic_and_monotonic():
|
||||||
|
"""Saubere Quoten → 10; jede zusätzliche Quote drückt die Note."""
|
||||||
|
sauber = {k: 0 for k in qa.NOTE_GEWICHTE}
|
||||||
|
assert qa.note(sauber) == 10.0
|
||||||
|
schlechter = dict(sauber, luecken=0.05)
|
||||||
|
noch_schlechter = dict(schlechter, fremd=0.05)
|
||||||
|
assert 10.0 > qa.note(schlechter) > qa.note(noch_schlechter) >= 0.0
|
||||||
|
assert qa.note({k: 1 for k in qa.NOTE_GEWICHTE}) == 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_note_kalibrierung():
|
||||||
|
"""Gewicht = Punktabzug bei 100 %: 5 % Fremd × 2.5 → −1.25 → 8.8 gerundet."""
|
||||||
|
assert qa.note({"fremd": 0.05}) == 8.8
|
||||||
|
assert qa.note({"fremd": 1.0}) == 0.0 # komplett fremdes Inventar = 0, nicht 7.7
|
||||||
|
|
||||||
|
|
||||||
|
def test_note_verdacht_zaehlt_nicht():
|
||||||
|
"""dubletten_verdacht ist Verdachtsliste, kein Urteil — beeinflusst die Note nicht."""
|
||||||
|
assert qa.note({"dubletten_verdacht": 1.0}) == 10.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_note_artefakte_getrennt():
|
||||||
|
"""Subs/Artefakte haben eigene Gewichte — zur Gate-Zeit existieren sie noch nicht
|
||||||
|
und dürfen die Inventar-Note weder schönen noch drücken."""
|
||||||
|
assert "subs_ohne_beleg" not in qa.NOTE_GEWICHTE
|
||||||
|
assert qa.note({"subs_ohne_beleg": 0.0, "verwaiste": 0.1}, qa.NOTE_GEWICHTE_ARTEFAKTE) == 9.0
|
||||||
|
assert qa.note({"subs_ohne_beleg": 1.0}, qa.NOTE_GEWICHTE_ARTEFAKTE) == 0.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_artefakte_coverage_and_orphans():
|
||||||
|
subs = [{"block_norm": "b", "sub_norm": "s1: lange beschreibung", "status": "consensus"},
|
||||||
|
{"block_norm": "b", "sub_norm": "s2", "status": "consensus"},
|
||||||
|
{"block_norm": "b", "sub_norm": "alt", "status": "variant"},
|
||||||
|
{"block_norm": "b", "sub_norm": "weg", "status": "discarded"}]
|
||||||
|
arts = [{"block_norm": "b", "sub_norm": "s1", "type": "flashcard"}, # Präfix-Treffer
|
||||||
|
{"block_norm": "b", "sub_norm": "alt", "type": "flashcard"}, # variant → lebt, keine Waise
|
||||||
|
{"block_norm": "b", "sub_norm": "weg", "type": "flashcard"}, # verworfen → Waise
|
||||||
|
{"block_norm": "b", "sub_norm": "tot", "type": "flashcard"}] # fehlt → Waise
|
||||||
|
fragen = [{"block_norm": "b", "sub_norm": "s1: lange beschreibung"},
|
||||||
|
{"block_norm": "b", "sub_norm": "s2"}]
|
||||||
|
r = qa.artefakte(subs, arts, fragen)
|
||||||
|
assert r["frage_abdeckung"] == 1.0
|
||||||
|
assert r["flashcard_abdeckung"] == 0.5 # nur s1 der beiden consensus-Subs
|
||||||
|
assert r["verwaiste"] == ["flashcard: b · tot", "flashcard: b · weg"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_artefakte_prefix_family_resolves_to_consensus():
|
||||||
|
"""Kurz-Key trifft consensus-Sub PLUS gefaltete Varianten mit gleichem Präfix —
|
||||||
|
das ist keine Waise, das Ziel ist der consensus-Sub."""
|
||||||
|
subs = [{"block_norm": "b", "sub_norm": "auto: echte fassung", "status": "consensus"},
|
||||||
|
{"block_norm": "b", "sub_norm": "auto: variante eins", "status": "variant"},
|
||||||
|
{"block_norm": "b", "sub_norm": "auto: variante zwei", "status": "variant"}]
|
||||||
|
arts = [{"block_norm": "b", "sub_norm": "auto", "type": "example"}]
|
||||||
|
r = qa.artefakte(subs, arts, [])
|
||||||
|
assert r["verwaiste"] == []
|
||||||
|
assert r["beispiel_abdeckung"] == 1.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_artefakte_not_generated():
|
||||||
|
assert qa.artefakte([{"block_norm": "b", "sub_norm": "s", "status": "consensus"}], [], []) == {"status": "nicht generiert"}
|
||||||
|
|
||||||
|
|
||||||
|
def test_fremd_glued_prefix_not_whitewashed():
|
||||||
|
"""'αÜbergang' darf nicht über den Substring 'bergang'⊂'Übergang' als belegt gelten;
|
||||||
|
Symbol-Varianten (Δ/∆) bleiben über die ASCII-Form gedeckt."""
|
||||||
|
corpus = {"S.txt": "Der Übergang ist wichtig.\n\nDer ∆TSP1 Algorithmus folgt."}
|
||||||
|
b = [{"title": "αÜbergang", "description": "", "sources": []},
|
||||||
|
{"title": "ΔTSP1-Algorithmus", "description": "", "sources": []}]
|
||||||
|
assert qa.fremd(b, corpus) == ["αÜbergang"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_note_ignores_unmeasured_quotes():
|
||||||
|
"""unechte_bloecke zählt nur, wenn gemessen (--llm) — sonst weder Schaden noch Schönung."""
|
||||||
|
ohne = {k: 0.02 for k in qa.NOTE_GEWICHTE if k != "unechte_bloecke"}
|
||||||
|
mit_null = dict(ohne, unechte_bloecke=0.0)
|
||||||
|
mit_schaden = dict(ohne, unechte_bloecke=0.5)
|
||||||
|
assert qa.note(mit_null) == qa.note(ohne)
|
||||||
|
assert qa.note(mit_schaden) < qa.note(ohne)
|
||||||
|
|
||||||
|
|
||||||
|
class _FakeEmb:
|
||||||
|
"""Gleicher Text → gleicher Einheitsvektor, sonst orthogonal (cos 1.0 / 0.0)."""
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def available():
|
||||||
|
return True
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def embed(texts):
|
||||||
|
import numpy as np
|
||||||
|
uniq = {t: k for k, t in enumerate(dict.fromkeys(texts))}
|
||||||
|
arr = np.zeros((len(texts), max(len(uniq), 1)))
|
||||||
|
for r, t in enumerate(texts):
|
||||||
|
arr[r, uniq[t]] = 1.0
|
||||||
|
return arr
|
||||||
|
|
||||||
|
|
||||||
|
def test_sub_dubletten_detector(monkeypatch):
|
||||||
|
"""Kandidaten in-block UND cross-block; nur consensus-Subs zählen."""
|
||||||
|
monkeypatch.setattr(qa, "embedding", _FakeEmb)
|
||||||
|
rows = [{"block": "Alpha", "block_norm": "alpha", "sub_title": "Gleiche Aussage", "status": "consensus"},
|
||||||
|
{"block": "Beta", "block_norm": "beta", "sub_title": "Gleiche Aussage", "status": "consensus"},
|
||||||
|
{"block": "Beta", "block_norm": "beta", "sub_title": "Andere Aussage", "status": "consensus"},
|
||||||
|
{"block": "Beta", "block_norm": "beta", "sub_title": "Gleiche Aussage", "status": "variant"}]
|
||||||
|
pairs = qa.sub_dubletten(rows)
|
||||||
|
assert len(pairs) == 1
|
||||||
|
assert pairs[0]["cross"] is True and pairs[0]["cos"] == 1.0
|
||||||
|
assert qa.sub_dubletten(rows, emb_on=False) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_note_sub_dubletten():
|
||||||
|
"""Bestätigte Sub-Dubletten drücken die Artefakt-Note; der bloße Verdacht nicht."""
|
||||||
|
assert qa.note({"sub_dubletten": 0.1}, qa.NOTE_GEWICHTE_ARTEFAKTE) == 9.0
|
||||||
|
assert qa.note({"sub_dubletten_verdacht": 1.0}, qa.NOTE_GEWICHTE_ARTEFAKTE) == 10.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_zaehlbare_luecken():
|
||||||
|
"""Mit LLM zählen widerlegte Lücken nicht; unbeurteilte ('?'/ohne Key) konservativ schon."""
|
||||||
|
lk = [{"llm": "ja"}, {"llm": "nein"}, {"llm": "?"}, {}]
|
||||||
|
assert len(qa._zaehlbare_luecken(lk, llm=True)) == 3
|
||||||
|
assert len(qa._zaehlbare_luecken(lk, llm=False)) == 4
|
||||||
|
|
||||||
|
|
||||||
|
def test_description_anchors_cover(monkeypatch):
|
||||||
|
"""Am Gate existieren keine Subs — Beschreibungs-Tokens müssen Abschnitte decken."""
|
||||||
|
monkeypatch.setattr(qa, "SECTION_CHARS", 20)
|
||||||
|
corpus = {"S.txt": "Kapitel 9: Augmentationswege und perfektes Matching."}
|
||||||
|
block = [{"title": "Paarungen", "description": "perfektes Matching mit Augmentationswege", "sources": []}]
|
||||||
|
assert qa.luecken(block, {}, corpus) == []
|
||||||
|
ohne = [{"title": "Paarungen", "description": "", "sources": []}]
|
||||||
|
assert len(qa.luecken(ohne, {}, corpus)) == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_fremd_digit_suffix_tolerant():
|
||||||
|
"""'ΔTSP1' matcht Korpus-'∆TSP' (tokenisiert zu 'tsp') via Ziffern-Suffix-Fallback."""
|
||||||
|
corpus = {"S.txt": "Der ∆TSP Algorithmus verdoppelt Kanten im Graphen."}
|
||||||
|
b = [{"title": "ΔTSP1-Algorithmus", "description": "", "sources": []},
|
||||||
|
{"title": "Quantencomputer", "description": "", "sources": []}]
|
||||||
|
assert qa.fremd(b, corpus) == ["Quantencomputer"]
|
||||||
230
backend/tests/test_repair.py
Normal file
230
backend/tests/test_repair.py
Normal file
@@ -0,0 +1,230 @@
|
|||||||
|
"""Befund-Repair: gezielte Aktionen aus dem QA-Report (repair.py) — ohne Flow, gegen Test-DB."""
|
||||||
|
|
||||||
|
import json
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
import repair
|
||||||
|
import qa as qa_mod
|
||||||
|
|
||||||
|
TOPIC = "reparatur"
|
||||||
|
|
||||||
|
|
||||||
|
def _report(**over):
|
||||||
|
r = {"topic": TOPIC, "note": 9.0, "quoten": {}, "hygiene": [], "dubletten": [],
|
||||||
|
"fremd": [], "unecht": [], "luecken": [], "artefakte": {"verwaiste": []}}
|
||||||
|
r.update(over)
|
||||||
|
return r
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
async def env(testdb, tmp_path, monkeypatch):
|
||||||
|
"""Zwei fertige Blöcke auf beiden Boards + Subs/Artefakte + Sidecar-Dateien."""
|
||||||
|
db = testdb
|
||||||
|
monkeypatch.setattr(qa_mod, "QA_DIR", tmp_path / "qa")
|
||||||
|
files = {"sidecar": tmp_path / "sidecar.json", "facts": tmp_path / "facts.json",
|
||||||
|
"question_pattern": tmp_path / "qp.json", "sub_roh": tmp_path / "roh.json",
|
||||||
|
"artefakte": tmp_path / "artefakte.json"}
|
||||||
|
monkeypatch.setattr(repair, "_blocks_files", lambda t: files)
|
||||||
|
# frisches Abschluss-QA im Repair stumm schalten (eigener Test deckt qa_report ab)
|
||||||
|
async def _no_qa(topic, llm=False):
|
||||||
|
return None
|
||||||
|
monkeypatch.setattr(qa_mod, "qa_report", _no_qa)
|
||||||
|
|
||||||
|
async def _seed(title, desc, subs=1):
|
||||||
|
norm = repair._norm_title(title)
|
||||||
|
cid = "b-" + norm.replace(" ", "")[:10]
|
||||||
|
await db.kanban_upsert_card(TOPIC, "inventory", cid, "block", "done_block",
|
||||||
|
{"title": title, "description": desc, "sources": [f"{title}.txt"],
|
||||||
|
"readers": ["r1"], "mirrored_norm": norm})
|
||||||
|
await db.kanban_upsert_card(TOPIC, "artefacts", norm, "ablock", "done_artefact", {"title": title})
|
||||||
|
await db.upsert_block(TOPIC, norm, title, desc, [f"{title}.txt"])
|
||||||
|
await db.set_block_status(TOPIC, norm, "consensus")
|
||||||
|
for i in range(subs):
|
||||||
|
await db.put_subblock(TOPIC, norm, f"sub{i}", title, f"Sub {i}")
|
||||||
|
await db.put_sub_artifact(TOPIC, norm, "sub0", "flashcard", "{}", title, "Sub 0")
|
||||||
|
return cid
|
||||||
|
|
||||||
|
for p in files.values():
|
||||||
|
p.write_text("{}", encoding="utf-8")
|
||||||
|
(tmp_path / "qa" / TOPIC).mkdir(parents=True)
|
||||||
|
|
||||||
|
def write_report(r):
|
||||||
|
(tmp_path / "qa" / TOPIC / "r.json").write_text(json.dumps(r), encoding="utf-8")
|
||||||
|
|
||||||
|
return db, _seed, files, write_report
|
||||||
|
|
||||||
|
|
||||||
|
async def test_merge_confirmed_duplicate(env, monkeypatch):
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
cid_a = await seed("Alpha", "kurz")
|
||||||
|
cid_b = await seed("Alpha Problem", "deutlich längere Beschreibung — Gewinner")
|
||||||
|
write_report(_report(dubletten=[{"a": "Alpha", "b": "Alpha Problem", "llm": "ja"},
|
||||||
|
{"a": "Alpha", "b": "Beta", "llm": "nein"}]))
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
async def fake_agent(key, prompt, timeout, **kw):
|
||||||
|
calls.append(prompt)
|
||||||
|
return 0, '{"relevant": {"1": "ja"}}', ""
|
||||||
|
|
||||||
|
monkeypatch.setattr(repair, "run_agent", fake_agent)
|
||||||
|
res = await repair.repair_befunde(TOPIC)
|
||||||
|
assert res["merges"] == ["Alpha → Alpha Problem"]
|
||||||
|
assert len(calls) == 1 and "Beta" not in calls[0] # nur das llm=ja-Paar zum Judge
|
||||||
|
verlierer = await db.kanban_get_card(TOPIC, "inventory", cid_a)
|
||||||
|
assert verlierer["stage"] == "grouped" and verlierer["payload"]["merged_into"] == "Alpha Problem"
|
||||||
|
gewinner = await db.kanban_get_card(TOPIC, "inventory", cid_b)
|
||||||
|
assert "Alpha.txt" in gewinner["payload"]["sources"] # Union
|
||||||
|
assert await db.kanban_get_card(TOPIC, "artefacts", "alpha") is None
|
||||||
|
assert not [r for r in await db.list_subblocks(TOPIC, "alpha")]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_fremd_removed_only_on_nein(env, monkeypatch):
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
cid_f = await seed("Fremdling", "gehört nicht rein")
|
||||||
|
cid_e = await seed("Echter", "belegt")
|
||||||
|
write_report(_report(fremd=["Fremdling", "Echter"]))
|
||||||
|
|
||||||
|
async def fake_agent(key, prompt, timeout, **kw):
|
||||||
|
return 0, '{"relevant": {"1": "nein", "2": "ja"}}', ""
|
||||||
|
|
||||||
|
monkeypatch.setattr(repair, "run_agent", fake_agent)
|
||||||
|
monkeypatch.setattr(repair, "source_folder", lambda t: None)
|
||||||
|
res = await repair.repair_befunde(TOPIC)
|
||||||
|
assert res["entfernt"] == ["Fremdling"]
|
||||||
|
weg = await db.kanban_get_card(TOPIC, "inventory", cid_f)
|
||||||
|
assert weg["stage"] == "rejected" and weg["payload"]["reason"] == "qa-fremd"
|
||||||
|
bleibt = await db.kanban_get_card(TOPIC, "inventory", cid_e)
|
||||||
|
assert bleibt["stage"] == "done_block"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_judge_failure_keeps_everything(env, monkeypatch):
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
cid = await seed("Wackelig", "unsicher")
|
||||||
|
write_report(_report(unecht=["Wackelig"]))
|
||||||
|
|
||||||
|
async def broken_agent(key, prompt, timeout, **kw):
|
||||||
|
raise RuntimeError("boom")
|
||||||
|
|
||||||
|
monkeypatch.setattr(repair, "run_agent", broken_agent)
|
||||||
|
res = await repair.repair_befunde(TOPIC)
|
||||||
|
assert res["entfernt"] == []
|
||||||
|
card = await db.kanban_get_card(TOPIC, "inventory", cid)
|
||||||
|
assert card["stage"] == "done_block" # fail-open
|
||||||
|
|
||||||
|
|
||||||
|
async def test_hygiene_cleans_title_norm_invariant(env, monkeypatch):
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
cid = await seed("**Fetter Titel**", "beschreibung")
|
||||||
|
files["sidecar"].write_text(json.dumps({"**Fetter Titel**": ["s"]}), encoding="utf-8")
|
||||||
|
write_report(_report(hygiene=[{"titel": "**Fetter Titel**", "probleme": ["markdown"]}]))
|
||||||
|
|
||||||
|
async def no_agent(*a, **kw):
|
||||||
|
raise AssertionError("Hygiene braucht keinen Agenten")
|
||||||
|
|
||||||
|
monkeypatch.setattr(repair, "run_agent", no_agent)
|
||||||
|
res = await repair.repair_befunde(TOPIC)
|
||||||
|
assert res["hygiene"] == ["**Fetter Titel** → Fetter Titel"]
|
||||||
|
card = await db.kanban_get_card(TOPIC, "inventory", cid)
|
||||||
|
assert card["payload"]["title"] == "Fetter Titel"
|
||||||
|
assert json.loads(files["sidecar"].read_text()) == {"Fetter Titel": ["s"]}
|
||||||
|
rows = await db.list_blocks(TOPIC)
|
||||||
|
assert any(r["title"] == "Fetter Titel" and r["status"] == "consensus" for r in rows)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_no_report_is_clean_error(env):
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
res = await repair.repair_befunde("gibtsnicht")
|
||||||
|
assert "fehler" in res
|
||||||
|
|
||||||
|
|
||||||
|
async def test_abschluss_qa_misst_mit_llm(env, monkeypatch):
|
||||||
|
"""Repair-Abschlussreport misst mit LLM — der llm=False-Report blendete
|
||||||
|
sub_dubletten aus und ließ die Note zwischen 10.0 und ~9 pendeln."""
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
write_report(_report())
|
||||||
|
import qa as qa_mod
|
||||||
|
seen = {}
|
||||||
|
|
||||||
|
async def spy(topic, llm=False):
|
||||||
|
seen["llm"] = llm
|
||||||
|
return None
|
||||||
|
|
||||||
|
monkeypatch.setattr(qa_mod, "qa_report", spy)
|
||||||
|
await repair.repair_befunde(TOPIC)
|
||||||
|
assert seen["llm"] is True
|
||||||
|
|
||||||
|
|
||||||
|
async def test_sub_dubletten_merge(env, monkeypatch):
|
||||||
|
"""Bestätigtes Sub-Paar + Zweitmeinung ja → Verlierer variant, Frage/Artefakt
|
||||||
|
wandern zum Gewinner (bzw. fallen weg, wenn er den Typ schon hat)."""
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
await seed("Alpha", "beschr")
|
||||||
|
norm = repair._norm_title("Alpha")
|
||||||
|
await db.put_subblock(TOPIC, norm, "gewinner sub", "Alpha", "Gewinner Sub",
|
||||||
|
facts='{"key_points": ["a", "b"]}', status="consensus")
|
||||||
|
await db.put_subblock(TOPIC, norm, "verlierer sub", "Alpha", "Verlierer Sub",
|
||||||
|
facts='{"key_points": ["x"]}', status="consensus")
|
||||||
|
await db.put_sub_artifact(TOPIC, norm, "verlierer sub", "example", "{}", "Alpha", "Verlierer Sub")
|
||||||
|
await db.upsert_question_pattern(TOPIC, norm, "verlierer sub", "Alpha", "Verlierer Sub", "Frage V?")
|
||||||
|
write_report(_report(sub_dubletten=[
|
||||||
|
{"a": "[Alpha] Gewinner Sub", "b": "[Alpha] Verlierer Sub", "llm": "ja"},
|
||||||
|
{"a": "[Alpha] Gibtsnicht", "b": "[Alpha] Verlierer Sub", "llm": "ja"}])) # tote Zeile → skip
|
||||||
|
|
||||||
|
async def fake_agent(key, prompt, timeout, **kw):
|
||||||
|
assert "Gibtsnicht" not in prompt
|
||||||
|
return 0, '{"relevant": {"1": "ja"}}', ""
|
||||||
|
|
||||||
|
monkeypatch.setattr(repair, "run_agent", fake_agent)
|
||||||
|
res = await repair.repair_befunde(TOPIC)
|
||||||
|
assert res["sub_merges"] == ["Verlierer Sub → Gewinner Sub"]
|
||||||
|
rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, norm)}
|
||||||
|
assert rows["verlierer sub"] == "variant" and rows["gewinner sub"] == "consensus"
|
||||||
|
arts = {(r["sub_norm"], r["type"]) for r in await db.get_sub_artefakte(TOPIC)}
|
||||||
|
assert ("gewinner sub", "example") in arts and ("verlierer sub", "example") not in arts
|
||||||
|
fragen = {r["sub_norm"]: r["question"] for r in await db.list_question_pattern(TOPIC)}
|
||||||
|
assert fragen.get("gewinner sub") == "Frage V?" and "verlierer sub" not in fragen
|
||||||
|
|
||||||
|
|
||||||
|
async def test_sub_dubletten_zweitmeinung_nein(env, monkeypatch):
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
await seed("Alpha", "beschr")
|
||||||
|
norm = repair._norm_title("Alpha")
|
||||||
|
await db.put_subblock(TOPIC, norm, "sub a", "Alpha", "Sub A", status="consensus")
|
||||||
|
await db.put_subblock(TOPIC, norm, "sub b", "Alpha", "Sub B", status="consensus")
|
||||||
|
write_report(_report(sub_dubletten=[{"a": "[Alpha] Sub A", "b": "[Alpha] Sub B", "llm": "ja"}]))
|
||||||
|
|
||||||
|
async def fake_agent(key, prompt, timeout, **kw):
|
||||||
|
return 0, '{"relevant": {"1": "nein"}}', ""
|
||||||
|
|
||||||
|
monkeypatch.setattr(repair, "run_agent", fake_agent)
|
||||||
|
res = await repair.repair_befunde(TOPIC)
|
||||||
|
assert res["sub_merges"] == []
|
||||||
|
rows = {r["sub_norm"]: r["status"] for r in await db.list_subblocks(TOPIC, norm)}
|
||||||
|
assert rows["sub a"] == rows["sub b"] == "consensus"
|
||||||
|
|
||||||
|
|
||||||
|
async def test_waisen_cleanup(env, monkeypatch):
|
||||||
|
"""Artefakte/Fragen auf verworfene oder fehlende Subs fliegen; lebende und
|
||||||
|
mehrdeutig-präfixige bleiben."""
|
||||||
|
db, seed, files, write_report = env
|
||||||
|
await seed("Alpha", "beschreibung") # legt sub0 (consensus) + flashcard auf sub0 an
|
||||||
|
norm = repair._norm_title("Alpha")
|
||||||
|
await db.put_subblock(TOPIC, norm, "weg", "Alpha", "Weg", status="discarded")
|
||||||
|
await db.put_subblock(TOPIC, norm, "doppel: eins", "Alpha", "Doppel eins")
|
||||||
|
await db.put_subblock(TOPIC, norm, "doppel: zwei", "Alpha", "Doppel zwei")
|
||||||
|
await db.put_sub_artifact(TOPIC, norm, "weg", "flashcard", "{}", "Alpha", "Weg") # tot
|
||||||
|
await db.put_sub_artifact(TOPIC, norm, "fehlt", "example", "{}", "Alpha", "Fehlt") # tot
|
||||||
|
await db.put_sub_artifact(TOPIC, norm, "doppel", "example", "{}", "Alpha", "Doppel") # mehrdeutig → bleibt
|
||||||
|
await db.upsert_question_pattern(TOPIC, norm, "fehlt", "Alpha", "Fehlt", "Frage?") # tot
|
||||||
|
write_report(_report())
|
||||||
|
|
||||||
|
async def no_agent(*a, **kw):
|
||||||
|
raise AssertionError("Aufräumen braucht keinen Agenten")
|
||||||
|
|
||||||
|
monkeypatch.setattr(repair, "run_agent", no_agent)
|
||||||
|
res = await repair.repair_befunde(TOPIC)
|
||||||
|
assert res["aufgeraeumt"] == 3
|
||||||
|
rest = {(r["sub_norm"], r["type"]) for r in await db.get_sub_artefakte(TOPIC)}
|
||||||
|
assert rest == {("sub0", "flashcard"), ("doppel", "example")}
|
||||||
|
assert not [r for r in await db.list_question_pattern(TOPIC)]
|
||||||
@@ -394,10 +394,12 @@ async def test_facts_check_inline_cited_evidence(sub_env, monkeypatch, tmp_path)
|
|||||||
"cited_facts": [{"text": "T", "source": "Skript.txt, Z.2"}],
|
"cited_facts": [{"text": "T", "source": "Skript.txt, Z.2"}],
|
||||||
"example_idea": ""}]}
|
"example_idea": ""}]}
|
||||||
|
|
||||||
|
sh = blx._subs_hash({"Alpha": ["Sub Eins"]}) # Resume-Dateien tragen den Sub-Satz-Hash
|
||||||
|
|
||||||
async def fake_slot(ctx2, label, *, key, prompt, role, capabilities, payload, timeout, on_line=None):
|
async def fake_slot(ctx2, label, *, key, prompt, role, capabilities, payload, timeout, on_line=None):
|
||||||
if "-facts-erg-" in key:
|
if "-facts-erg-" in key:
|
||||||
return blx.FAILED, None
|
return blx.FAILED, None
|
||||||
(tmp_path / "facts-c0.json").write_text(json.dumps(facts), encoding="utf-8")
|
(tmp_path / f"facts-{sh}-c0.json").write_text(json.dumps(facts), encoding="utf-8")
|
||||||
return blx.OK, None
|
return blx.OK, None
|
||||||
|
|
||||||
seen = []
|
seen = []
|
||||||
@@ -416,4 +418,29 @@ async def test_facts_check_inline_cited_evidence(sub_env, monkeypatch, tmp_path)
|
|||||||
assert len(seen) == blx.FACTS_CHECK_PANEL
|
assert len(seen) == blx.FACTS_CHECK_PANEL
|
||||||
key, caps, prompt = seen[0]
|
key, caps, prompt = seen[0]
|
||||||
assert caps == "none" and "── Skript.txt · Z." in prompt
|
assert caps == "none" and "── Skript.txt · Z." in prompt
|
||||||
assert (tmp_path / "facts-check-c0-j1.json").exists() # Engine persistiert die Antwort
|
assert (tmp_path / f"facts-check-{sh}-c0-j1.json").exists() # Engine persistiert die Antwort
|
||||||
|
|
||||||
|
|
||||||
|
def test_sub_key_resolves_short_titles():
|
||||||
|
"""Artefakt-Agenten echoen den Kurztitel; der Sub-Key heißt 'kurztitel: beschreibung'.
|
||||||
|
Eindeutiger Präfix wird aufgelöst, Mehrdeutiges und Fehlendes bleibt unverändert."""
|
||||||
|
import board_artefacts as ba
|
||||||
|
existing = {"autolink mit url: erzeugt link", "bilder: bindet bilder ein",
|
||||||
|
"doppel: eins", "doppel: zwei", "exakt"}
|
||||||
|
assert ba._sub_key(existing, "exakt") == "exakt"
|
||||||
|
assert ba._sub_key(existing, "autolink mit url") == "autolink mit url: erzeugt link"
|
||||||
|
assert ba._sub_key(existing, "doppel") == "doppel" # mehrdeutig → unverändert
|
||||||
|
assert ba._sub_key(existing, "fehlt") == "fehlt" # kein Treffer → unverändert
|
||||||
|
# Fuzzy: Paraphrase/Kürzung ohne Doppelpunkt-Präfix löst eindeutig auf
|
||||||
|
lang = {"der backslash selbst muss mit escaped werden, um literal zu erscheinen"}
|
||||||
|
assert ba._sub_key(lang, "der backslash selbst muss mit escaped werden") == next(iter(lang))
|
||||||
|
assert ba._sub_key(lang | {"der backslash am zeilenende"}, "der backslash") == "der backslash" # mehrdeutig
|
||||||
|
|
||||||
|
|
||||||
|
def test_subs_hash_invalidiert_bei_neuem_zuschnitt():
|
||||||
|
"""Gleicher Sub-Satz → gleicher Hash (Resume greift); geänderter → neuer Hash.
|
||||||
|
raw-Form (Strings) und sidecar-Form (dicts) hashen identisch."""
|
||||||
|
a = {"Block": ["s1", "s2"]}
|
||||||
|
assert blx._subs_hash(a) == blx._subs_hash({"Block": ["s1", "s2"]})
|
||||||
|
assert blx._subs_hash(a) != blx._subs_hash({"Block": ["s1", "s3"]})
|
||||||
|
assert blx._subs_hash(a) == blx._subs_hash({"Block": [{"title": "s1"}, {"title": "s2"}]})
|
||||||
|
|||||||
@@ -179,7 +179,7 @@ function selectTopic(topic) {
|
|||||||
selectedTopic.value = topic
|
selectedTopic.value = topic
|
||||||
previewGuide.value = null
|
previewGuide.value = null
|
||||||
sidebarSticky.value = false
|
sidebarSticky.value = false
|
||||||
mainView.value = 'blocks' // topic click → blocks overview (guide only on pill click)
|
mainView.value = 'generation' // topic click → generation board (guide only on pill click)
|
||||||
viewMode.value = localStorage.getItem('ansicht_' + topic) === 'erklärend' ? 'erklärend' : 'compact'
|
viewMode.value = localStorage.getItem('ansicht_' + topic) === 'erklärend' ? 'erklärend' : 'compact'
|
||||||
localStorage.setItem('lastTopic', topic)
|
localStorage.setItem('lastTopic', topic)
|
||||||
loadBlocks()
|
loadBlocks()
|
||||||
@@ -237,12 +237,12 @@ async function handleRequeueDead() {
|
|||||||
startPolling()
|
startPolling()
|
||||||
}
|
}
|
||||||
|
|
||||||
async function handleBlocksClick({ instructions = '', research = true }) {
|
async function handleBlocksClick({ instructions = '', research = true, qaForce = false }) {
|
||||||
if (!selectedTopic.value) return
|
if (!selectedTopic.value) return
|
||||||
uiError.value = null
|
uiError.value = null
|
||||||
try {
|
try {
|
||||||
// research=true = Start/mehr Research anhängen; false = Continue (Queue abarbeiten).
|
// research=true = Start/mehr Research anhängen; false = Continue (Queue abarbeiten).
|
||||||
await apiCreateBausteine(selectedTopic.value, instructions, provider.value, undefined, undefined, research)
|
await apiCreateBausteine(selectedTopic.value, instructions, provider.value, undefined, undefined, research, qaForce)
|
||||||
} catch (e) {
|
} catch (e) {
|
||||||
uiError.value = e.message
|
uiError.value = e.message
|
||||||
return
|
return
|
||||||
@@ -501,7 +501,7 @@ onMounted(async () => {
|
|||||||
@close="mainView = 'blocks'"
|
@close="mainView = 'blocks'"
|
||||||
@resetStage="handleResetStage"
|
@resetStage="handleResetStage"
|
||||||
@restartAll="() => handleBlocksClick({ research: true })"
|
@restartAll="() => handleBlocksClick({ research: true })"
|
||||||
@continueAll="() => handleBlocksClick({ research: false })"
|
@continueAll="(opts) => handleBlocksClick({ research: false, qaForce: !!(opts && opts.qaForce) })"
|
||||||
@addResearch="handleAddResearch"
|
@addResearch="handleAddResearch"
|
||||||
@requeueDead="handleRequeueDead"
|
@requeueDead="handleRequeueDead"
|
||||||
@removeAll="handleResetBlocks"
|
@removeAll="handleResetBlocks"
|
||||||
|
|||||||
@@ -37,11 +37,11 @@ export async function fetchBlocksStatus(topic) {
|
|||||||
return res.json()
|
return res.json()
|
||||||
}
|
}
|
||||||
|
|
||||||
export async function createBlocks(topic, instructions = '', provider = 'claude', sourceType = 'thema', sourceOrt = '', research = true) {
|
export async function createBlocks(topic, instructions = '', provider = 'claude', sourceType = 'thema', sourceOrt = '', research = true, qaForce = false) {
|
||||||
const res = await fetch(`${BASE}/blocks`, {
|
const res = await fetch(`${BASE}/blocks`, {
|
||||||
method: 'POST',
|
method: 'POST',
|
||||||
headers: { 'Content-Type': 'application/json' },
|
headers: { 'Content-Type': 'application/json' },
|
||||||
body: JSON.stringify({ topic, instructions, provider, source_type: sourceType, source_location: sourceOrt, research }),
|
body: JSON.stringify({ topic, instructions, provider, source_type: sourceType, source_location: sourceOrt, research, qa_force: qaForce }),
|
||||||
})
|
})
|
||||||
return jsonOrThrow(res)
|
return jsonOrThrow(res)
|
||||||
}
|
}
|
||||||
@@ -52,6 +52,26 @@ export async function fetchBlocksBoard(topic) {
|
|||||||
return jsonOrThrow(res)
|
return jsonOrThrow(res)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Manueller QA-Lauf (wie das Gate, inkl. LLM-Stichprobe); Badge liest den neuen Report.
|
||||||
|
export async function runQa(topic, llm = true) {
|
||||||
|
const res = await fetch(`${BASE}/blocks/qa`, {
|
||||||
|
method: 'POST',
|
||||||
|
headers: { 'Content-Type': 'application/json' },
|
||||||
|
body: JSON.stringify({ topic, llm }),
|
||||||
|
})
|
||||||
|
return jsonOrThrow(res)
|
||||||
|
}
|
||||||
|
|
||||||
|
// QA-Befunde gezielt beheben (Hygiene, bestätigte Dubletten, Fremd/Unecht nach Gegen-Judge).
|
||||||
|
export async function runRepair(topic) {
|
||||||
|
const res = await fetch(`${BASE}/blocks/repair`, {
|
||||||
|
method: 'POST',
|
||||||
|
headers: { 'Content-Type': 'application/json' },
|
||||||
|
body: JSON.stringify({ topic }),
|
||||||
|
})
|
||||||
|
return jsonOrThrow(res)
|
||||||
|
}
|
||||||
|
|
||||||
// Karten ab Spalte zurücksetzen (keine Generierung).
|
// Karten ab Spalte zurücksetzen (keine Generierung).
|
||||||
export async function resetBlocksStage(topic, board, stage) {
|
export async function resetBlocksStage(topic, board, stage) {
|
||||||
const res = await fetch(`${BASE}/blocks/reset-stage`, {
|
const res = await fetch(`${BASE}/blocks/reset-stage`, {
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
<script setup>
|
<script setup>
|
||||||
import { ref, computed, watch, onUnmounted } from 'vue'
|
import { ref, computed, watch, onUnmounted } from 'vue'
|
||||||
import { fetchBlocksBoard } from '../api.js'
|
import { fetchBlocksBoard, runQa, runRepair } from '../api.js'
|
||||||
import KanbanBoard from './KanbanBoard.vue'
|
import KanbanBoard from './KanbanBoard.vue'
|
||||||
import GuideBoardSection from './GuideBoardSection.vue'
|
import GuideBoardSection from './GuideBoardSection.vue'
|
||||||
|
|
||||||
@@ -36,13 +36,13 @@ onUnmounted(stopPoll)
|
|||||||
|
|
||||||
const inventoryCols = computed(() => (board.value?.columns || []).filter((c) => c.board === 'inventory'))
|
const inventoryCols = computed(() => (board.value?.columns || []).filter((c) => c.board === 'inventory'))
|
||||||
const artefactCols = computed(() => (board.value?.columns || []).filter((c) => c.board === 'artefacts'))
|
const artefactCols = computed(() => (board.value?.columns || []).filter((c) => c.board === 'artefacts'))
|
||||||
const boardEmpty = computed(() => !(board.value?.columns || []).some((c) => c.total > 0))
|
|
||||||
const dead = computed(() => board.value?.dead || [])
|
const dead = computed(() => board.value?.dead || [])
|
||||||
|
const qa = computed(() => board.value?.qa || null)
|
||||||
|
|
||||||
// Spalten, auf die zurückgesetzt werden kann (Terminal-Spalten sind kein Reset-Ziel).
|
// Spalten, auf die zurückgesetzt werden kann (Terminal-Spalten sind kein Reset-Ziel).
|
||||||
const RESETTABLE = new Set(['ingest', 'cluster', 'pair_check', 'consensus_gate', 'clarify', 'naming',
|
const RESETTABLE = new Set(['ingest', 'cluster', 'pair_check', 'consensus_gate', 'clarify', 'naming',
|
||||||
'naming_check', 'fragment_filter', 'grouping', 'gap_check', 'done',
|
'naming_check', 'fragment_filter', 'grouping', 'gap_check', 'done',
|
||||||
'subblocks', 'facts', 'levels', 'relevance', 'question_pattern', 'artefacts', 'finalize', 'outline'])
|
'subblocks', 'facts', 'konsolidierung', 'levels', 'relevance', 'question_pattern', 'artefacts', 'finalize', 'outline'])
|
||||||
const sel = ref(null) // gewählte Spalte {board, key, label}
|
const sel = ref(null) // gewählte Spalte {board, key, label}
|
||||||
const selCard = ref(null) // gewählte Karte (Einzel-Restart, nur artefacts)
|
const selCard = ref(null) // gewählte Karte (Einzel-Restart, nur artefacts)
|
||||||
const confirm = ref(null) // 2-Klick-Bestätigung für destruktive Aktionen
|
const confirm = ref(null) // 2-Klick-Bestätigung für destruktive Aktionen
|
||||||
@@ -82,6 +82,39 @@ function resetHere(restart) {
|
|||||||
confirm.value = null
|
confirm.value = null
|
||||||
later(() => emit('resetStage', { board: s.board, stage: s.key, restart }))
|
later(() => emit('resetStage', { board: s.board, stage: s.key, restart }))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const qaBusy = ref(false)
|
||||||
|
async function runQaClick() {
|
||||||
|
if (qaBusy.value) return
|
||||||
|
qaBusy.value = true
|
||||||
|
try {
|
||||||
|
await runQa(props.topic)
|
||||||
|
} finally {
|
||||||
|
qaBusy.value = false
|
||||||
|
pollBoard()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const repairBusy = ref(false)
|
||||||
|
const repairInfo = ref('')
|
||||||
|
async function repairClick() {
|
||||||
|
if (repairBusy.value) return
|
||||||
|
repairBusy.value = true
|
||||||
|
repairInfo.value = ''
|
||||||
|
try {
|
||||||
|
const r = await runRepair(props.topic)
|
||||||
|
const n = (r.hygiene || []).length + (r.merges || []).length + (r.sub_merges || []).length
|
||||||
|
+ (r.entfernt || []).length + (r.aufgeraeumt || 0)
|
||||||
|
repairInfo.value = n === 0
|
||||||
|
? 'keine behebbaren Befunde'
|
||||||
|
: `${(r.hygiene || []).length} Titel · ${(r.merges || []).length} Merges · ${(r.sub_merges || []).length} Sub-Merges · ${(r.entfernt || []).length} entfernt · ${r.aufgeraeumt || 0} aufgeräumt`
|
||||||
|
} catch (e) {
|
||||||
|
repairInfo.value = String(e.message || e)
|
||||||
|
} finally {
|
||||||
|
repairBusy.value = false
|
||||||
|
pollBoard()
|
||||||
|
}
|
||||||
|
}
|
||||||
</script>
|
</script>
|
||||||
|
|
||||||
<template>
|
<template>
|
||||||
@@ -93,11 +126,22 @@ function resetHere(restart) {
|
|||||||
<button class="gen-close" title="Close" @click="emit('close')">✕</button>
|
<button class="gen-close" title="Close" @click="emit('close')">✕</button>
|
||||||
</header>
|
</header>
|
||||||
|
|
||||||
|
<div v-if="qa && qa.pausiert" class="qa-pause">
|
||||||
|
<strong>QA-Gate: Note {{ qa.note.toFixed(1) }} unter Schwelle {{ qa.schwelle }} — pausiert.</strong>
|
||||||
|
<span v-if="qa.befunde.length"> Befunde: {{ qa.befunde.join(' · ') }}</span>
|
||||||
|
<button class="gen-act" @click="emit('continueAll', { qaForce: true })">Trotzdem fortsetzen</button>
|
||||||
|
</div>
|
||||||
<section class="gen-section">
|
<section class="gen-section">
|
||||||
<div class="gen-steps-top">
|
<div class="gen-steps-top">
|
||||||
<span class="gen-title">Bausteine</span>
|
<span class="gen-title">Bausteine</span>
|
||||||
<div v-if="progress" class="gen-progress"><span class="gen-progress-dot"></span>{{ progress }}</div>
|
<div v-if="progress" class="gen-progress"><span class="gen-progress-dot"></span>{{ progress }}</div>
|
||||||
<div v-if="!generating" class="gen-actions">
|
<div v-if="!generating" class="gen-actions">
|
||||||
|
<button class="gen-act" :disabled="qaBusy" title="QA-Lauf wie am Gate (inkl. LLM-Stichprobe)"
|
||||||
|
@click="runQaClick">{{ qaBusy ? 'QA läuft…' : 'QA' }}</button>
|
||||||
|
<button v-if="qa" class="gen-act" :disabled="repairBusy"
|
||||||
|
title="QA-Befunde gezielt beheben: Hygiene, bestätigte Dubletten mergen, Fremd/Unecht nach Gegen-Judge entfernen"
|
||||||
|
@click="repairClick">{{ repairBusy ? 'Repariert…' : 'Befunde beheben' }}</button>
|
||||||
|
<span v-if="repairInfo" class="repair-info">{{ repairInfo }}</span>
|
||||||
<button class="gen-act play" @click="emit('restartAll')">{{ ready || partial ? '+ Research' : 'Generate' }}</button>
|
<button class="gen-act play" @click="emit('restartAll')">{{ ready || partial ? '+ Research' : 'Generate' }}</button>
|
||||||
<button v-if="partial" class="gen-act" @click="emit('continueAll')">Continue</button>
|
<button v-if="partial" class="gen-act" @click="emit('continueAll')">Continue</button>
|
||||||
<button
|
<button
|
||||||
@@ -119,28 +163,34 @@ function resetHere(restart) {
|
|||||||
>⟳ {{ dead.length }} dead</button>
|
>⟳ {{ dead.length }} dead</button>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
<div v-if="boardEmpty && !generating" class="gen-empty">No board yet — generation streams live cards through the columns here.</div>
|
<div class="gen-board-label">
|
||||||
<template v-else>
|
Inventar
|
||||||
<div class="gen-board-label">Inventar</div>
|
<span v-if="qa" class="qa-note" :class="qa.note >= qa.schwelle ? 'ok' : 'bad'"
|
||||||
<KanbanBoard
|
:title="'QA-Schwelle ' + qa.schwelle">QA {{ qa.note.toFixed(1) }}/10</span>
|
||||||
:columns="inventoryCols"
|
</div>
|
||||||
:agents="board?.agents || []"
|
<KanbanBoard
|
||||||
:generating="generating"
|
:columns="inventoryCols"
|
||||||
:selectable="!generating"
|
:agents="board?.agents || []"
|
||||||
:selectedKey="sel?.board === 'inventory' ? sel.key : null"
|
:generating="generating"
|
||||||
@stageClick="stageClick"
|
:selectable="!generating"
|
||||||
/>
|
:selectedKey="sel?.board === 'inventory' ? sel.key : null"
|
||||||
<div class="gen-board-label">Artefakte</div>
|
@stageClick="stageClick"
|
||||||
<KanbanBoard
|
/>
|
||||||
:columns="artefactCols"
|
<div class="gen-board-label">
|
||||||
:generating="generating"
|
Artefakte
|
||||||
:selectable="!generating"
|
<span v-if="qa && qa.note_artefakte != null" class="qa-note"
|
||||||
:cardSelectable="!generating"
|
:class="qa.note_artefakte >= qa.schwelle ? 'ok' : 'bad'"
|
||||||
:selectedKey="sel?.board === 'artefacts' ? sel.key : null"
|
title="Beleg-Quote + verwaiste Artefakte">QA {{ qa.note_artefakte.toFixed(1) }}/10</span>
|
||||||
@stageClick="stageClick"
|
</div>
|
||||||
@cardClick="cardClick"
|
<KanbanBoard
|
||||||
/>
|
:columns="artefactCols"
|
||||||
</template>
|
:generating="generating"
|
||||||
|
:selectable="!generating"
|
||||||
|
:cardSelectable="!generating"
|
||||||
|
:selectedKey="sel?.board === 'artefacts' ? sel.key : null"
|
||||||
|
@stageClick="stageClick"
|
||||||
|
@cardClick="cardClick"
|
||||||
|
/>
|
||||||
|
|
||||||
<div v-if="selCard && !generating" class="gen-step-actions">
|
<div v-if="selCard && !generating" class="gen-step-actions">
|
||||||
<span class="gen-step-actions-label">Karte «{{ selCard.title }}»:</span>
|
<span class="gen-step-actions-label">Karte «{{ selCard.title }}»:</span>
|
||||||
@@ -220,7 +270,6 @@ function resetHere(restart) {
|
|||||||
color: var(--text-faint);
|
color: var(--text-faint);
|
||||||
margin: 0.5rem 0 0.3rem;
|
margin: 0.5rem 0 0.3rem;
|
||||||
}
|
}
|
||||||
.gen-empty { color: var(--text-faint); font-size: 0.82rem; padding: 0.4rem 0; }
|
|
||||||
.gen-progress {
|
.gen-progress {
|
||||||
display: flex;
|
display: flex;
|
||||||
align-items: center;
|
align-items: center;
|
||||||
@@ -267,4 +316,25 @@ function resetHere(restart) {
|
|||||||
.gen-act.danger { color: var(--danger); border-color: var(--danger); background: transparent; }
|
.gen-act.danger { color: var(--danger); border-color: var(--danger); background: transparent; }
|
||||||
.gen-act.danger.armed { background: var(--danger); color: #fff; }
|
.gen-act.danger.armed { background: var(--danger); color: #fff; }
|
||||||
.gen-act.ghost { color: var(--text-muted); }
|
.gen-act.ghost { color: var(--text-muted); }
|
||||||
|
.qa-note {
|
||||||
|
font-size: 0.78rem;
|
||||||
|
font-weight: 600;
|
||||||
|
padding: 0.1rem 0.5rem;
|
||||||
|
border-radius: 999px;
|
||||||
|
}
|
||||||
|
.qa-note.ok { background: color-mix(in srgb, #22c55e 18%, transparent); color: #16a34a; }
|
||||||
|
.qa-note.bad { background: color-mix(in srgb, #ef4444 18%, transparent); color: #dc2626; }
|
||||||
|
.repair-info { font-size: 0.78rem; color: var(--text-muted); }
|
||||||
|
.qa-pause {
|
||||||
|
display: flex;
|
||||||
|
align-items: center;
|
||||||
|
gap: 0.6rem;
|
||||||
|
flex-wrap: wrap;
|
||||||
|
padding: 0.5rem 0.8rem;
|
||||||
|
margin-bottom: 0.8rem;
|
||||||
|
border: 1px solid color-mix(in srgb, #ef4444 40%, transparent);
|
||||||
|
border-radius: 8px;
|
||||||
|
background: color-mix(in srgb, #ef4444 8%, transparent);
|
||||||
|
font-size: 0.85rem;
|
||||||
|
}
|
||||||
</style>
|
</style>
|
||||||
|
|||||||
@@ -79,6 +79,8 @@ function resetHere() {
|
|||||||
<section class="gb-board">
|
<section class="gb-board">
|
||||||
<div class="gb-top">
|
<div class="gb-top">
|
||||||
<span class="gb-title">Guide · {{ format }}</span>
|
<span class="gb-title">Guide · {{ format }}</span>
|
||||||
|
<span v-if="board?.qa_guide != null" class="gb-qa" :class="board.qa_guide >= 9 ? 'ok' : 'bad'"
|
||||||
|
title="Guide-QA (make qa-guide)">QA {{ board.qa_guide.toFixed(1) }}/10</span>
|
||||||
<span v-if="total" class="gb-count">{{ done }}/{{ total }} Karten fertig</span>
|
<span v-if="total" class="gb-count">{{ done }}/{{ total }} Karten fertig</span>
|
||||||
<div v-if="board?.progress && generating" class="gb-progress"><span class="gb-progress-dot"></span>{{ board.progress }}</div>
|
<div v-if="board?.progress && generating" class="gb-progress"><span class="gb-progress-dot"></span>{{ board.progress }}</div>
|
||||||
<div v-if="board?.error" class="gb-error">{{ board.error }}</div>
|
<div v-if="board?.error" class="gb-error">{{ board.error }}</div>
|
||||||
@@ -126,6 +128,14 @@ function resetHere() {
|
|||||||
.gb-top { display: flex; align-items: center; gap: 1rem; min-height: 1.9rem; margin-bottom: 0.6rem; }
|
.gb-top { display: flex; align-items: center; gap: 1rem; min-height: 1.9rem; margin-bottom: 0.6rem; }
|
||||||
.gb-title { font-size: 0.9rem; font-weight: 700; }
|
.gb-title { font-size: 0.9rem; font-weight: 700; }
|
||||||
.gb-count { color: var(--text-muted); font-size: 0.82rem; }
|
.gb-count { color: var(--text-muted); font-size: 0.82rem; }
|
||||||
|
.gb-qa {
|
||||||
|
font-size: 0.78rem;
|
||||||
|
font-weight: 600;
|
||||||
|
padding: 0.1rem 0.5rem;
|
||||||
|
border-radius: 999px;
|
||||||
|
}
|
||||||
|
.gb-qa.ok { background: color-mix(in srgb, #22c55e 18%, transparent); color: #16a34a; }
|
||||||
|
.gb-qa.bad { background: color-mix(in srgb, #ef4444 18%, transparent); color: #dc2626; }
|
||||||
.gb-progress {
|
.gb-progress {
|
||||||
display: flex;
|
display: flex;
|
||||||
align-items: center;
|
align-items: center;
|
||||||
|
|||||||
20
templates/Prompt/Blocks-Supplement-Material.md
Normal file
20
templates/Prompt/Blocks-Supplement-Material.md
Normal file
@@ -0,0 +1,20 @@
|
|||||||
|
Check the block inventory for the topic "{topic}" for completeness against the LEARNING MATERIAL — the material defines the scope, nothing else.
|
||||||
|
|
||||||
|
The material is in the folder {project} (PDFs are provided as same-named .txt files — ALWAYS read the .txt, never the PDF). Get an overview with Bash (ls/find), read the files, then compare: which distinct concepts, methods, algorithms or theorems does the MATERIAL treat that the inventory misses?
|
||||||
|
|
||||||
|
EXISTING BLOCKS:
|
||||||
|
{blocks}
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Propose ONLY blocks the material itself treats (a heading, definition, algorithm, proof or exercise about it) and that are missing from the inventory.
|
||||||
|
- Do NOT search the web. Do NOT propose textbook canon beyond the material.
|
||||||
|
- A block solves EXACTLY ONE PROBLEM and is ATOMIC — same standards as the inventory.
|
||||||
|
- NO variants, rephrasings, or deep-dives of existing blocks — only genuine gaps.
|
||||||
|
- Title and description in GERMAN (technical terms stay original), description at most ~12 words.
|
||||||
|
- If there are no gaps, return an empty list — that is a valid result.
|
||||||
|
|
||||||
|
Write ONLY the JSON file to: {out_path}
|
||||||
|
|
||||||
|
Format:
|
||||||
|
{{"blocks": [{{"title": "…", "description": "…"}}]}}
|
||||||
|
No gaps: {{"blocks": []}}
|
||||||
@@ -9,9 +9,10 @@ VERIFIED FACTS — the ONLY allowed factual basis (extract-once from the source)
|
|||||||
Procedure:
|
Procedure:
|
||||||
1. Decompose the section text into its ATOMIC factual claims (definitions, properties, numbers, procedure steps, causal statements). Ignore pure didactics (transitions, framing, mnemonic phrasing).
|
1. Decompose the section text into its ATOMIC factual claims (definitions, properties, numbers, procedure steps, causal statements). Ignore pure didactics (transitions, framing, mnemonic phrasing).
|
||||||
2. Worked-example passages (a concrete problem worked through in steps to a result) are DIDACTICS when they merely APPLY or ILLUSTRATE a verified fact or a provided worked example: their concretely chosen values and computed intermediates do NOT count as over-specific. Flag an example step ONLY if it contradicts the facts or smuggles in a NEW general claim not derivable from them.
|
2. Worked-example passages (a concrete problem worked through in steps to a result) are DIDACTICS when they merely APPLY or ILLUSTRATE a verified fact or a provided worked example: their concretely chosen values and computed intermediates do NOT count as over-specific. Flag an example step ONLY if it contradicts the facts or smuggles in a NEW general claim not derivable from them.
|
||||||
3. Check EACH claim INDIVIDUALLY and BINARY against the VERIFIED FACTS above: derivable from them → belegt. Not derivable, contradicting, or over-specific beyond the facts → nicht belegt.
|
3. CONTEXT sentences are DIDACTICS too, not claims: introductions and summaries that only preview/recap the section, uncontroversial general knowledge that merely places the topic (history, origin, what an adjacent well-known technology is), and paraphrases of the verified facts. Ignore them — a guide needs connective tissue. This exemption ends the moment a sentence makes a checkable statement about THIS block's own syntax, behavior or rules: that is a claim.
|
||||||
4. Do NOT search the web, do NOT use outside knowledge — a claim that is true in the world but absent from the facts is still "nicht belegt".
|
4. Check EACH claim INDIVIDUALLY and BINARY against the VERIFIED FACTS above: derivable from them → belegt. Not derivable, contradicting, or over-specific beyond the facts → nicht belegt.
|
||||||
5. When in doubt → nicht belegt (the guide may only teach verified material).
|
5. Do NOT search the web, do NOT use outside knowledge as EVIDENCE — a claim about the block that is true in the world but absent from the facts is still "nicht belegt".
|
||||||
|
6. When a sentence is genuinely ambiguous between context and claim → treat it as a claim (safety before cost).
|
||||||
|
|
||||||
Write ONLY the JSON file to: {out_path}
|
Write ONLY the JSON file to: {out_path}
|
||||||
|
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ Review each section:
|
|||||||
|
|
||||||
You only REVIEW and note problems — you change nothing. Note only genuine flaws, no matters of taste.
|
You only REVIEW and note problems — you change nothing. Note only genuine flaws, no matters of taste.
|
||||||
|
|
||||||
Write ONLY the JSON file to: {out_path}
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
|
||||||
Format — all in order:
|
Format — all in order:
|
||||||
{{"ok": true}}
|
{{"ok": true}}
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ HOW TO WRITE the detailed version — ONE coherent text for a junior who is lear
|
|||||||
- Resolve EVERY technical term at first mention in half a sentence. Use "Prerequisite" hints as anchors, "Hurdle" hints to clear misconceptions up front.
|
- Resolve EVERY technical term at first mention in half a sentence. Use "Prerequisite" hints as anchors, "Hurdle" hints to clear misconceptions up front.
|
||||||
- Order of the subblocks as in the assignment: first `[beginner]`, then `[advanced]`, then `[expert]`, last `[peripheral]`.
|
- Order of the subblocks as in the assignment: first `[beginner]`, then `[advanced]`, then `[expert]`, last `[peripheral]`.
|
||||||
- The NUMBER of subblocks is the depth signal: many subblocks → the block deserves length; few → keep it short. Never pad.
|
- The NUMBER of subblocks is the depth signal: many subblocks → the block deserves length; few → keep it short. Never pad.
|
||||||
|
- LENGTH TARGET for the detailed version: about {budget} characters total (guideline, not a hard limit — covering ALL learning objectives always beats brevity). Unguided sections came out 2–4× too long and had to be rewritten.
|
||||||
- CONCISE: every sentence carries new information. No repetition, no filler, no preamble. When in doubt, leave it out.
|
- CONCISE: every sentence carries new information. No repetition, no filler, no preamble. When in doubt, leave it out.
|
||||||
- Show "how" procedures step by step. An example ONLY where it genuinely carries the understanding.
|
- Show "how" procedures step by step. An example ONLY where it genuinely carries the understanding.
|
||||||
- If VERIFIED WORKED EXAMPLES are provided above, weave each into its subblock as instructed there; do NOT invent additional worked examples for subblocks that already have one.
|
- If VERIFIED WORKED EXAMPLES are provided above, weave each into its subblock as instructed there; do NOT invent additional worked examples for subblocks that already have one.
|
||||||
|
|||||||
@@ -11,8 +11,10 @@ Relevance:
|
|||||||
|
|
||||||
Rules:
|
Rules:
|
||||||
- These subblocks **all already** belong to the topic — the question is not *whether* but *how central*. Judge **core vs. peripheral within its own block**, one at a time.
|
- These subblocks **all already** belong to the topic — the question is not *whether* but *how central*. Judge **core vs. peripheral within its own block**, one at a time.
|
||||||
- `peripheral` is a **genuine category**, not a leftover bucket. Deliberately mark peripheral items (nice-to-know, special/niche cases, detail/numeric examples, historical notes, cross-references) as `peripheral`.
|
- **The basic syntax / core rule of a block — what the block TITLE promises — is NEVER peripheral.** A block "Headings" without its heading syntax marked relevant is broken. Check this per block before anything else.
|
||||||
- Don't bump the level up out of caution. Only **true core concepts / central theorems** are never `peripheral`.
|
- **When the BLOCK itself is a fringe/extension feature** (non-standard, niche, an add-on rather than core of the topic), only its basic syntax and purpose are relevant — its detail rules and special cases are `peripheral`. A fringe feature must not be decomposed deeper than the core topics (measured: a non-standard block carried 22 "relevant" detail rules while core blocks had 6).
|
||||||
|
- `peripheral` is a **genuine category** for clear cases: nice-to-know, special/niche cases, detail/numeric examples, historical notes, cross-references, style tips.
|
||||||
|
- `peripheral` items are CUT from the focused guides — a wrong `peripheral` loses content, a wrong `relevant` only adds a paragraph. **When in doubt → relevant.**
|
||||||
- Only judge — invent nothing, change no subblocks.
|
- Only judge — invent nothing, change no subblocks.
|
||||||
|
|
||||||
Reply with ONLY the JSON — no other text, no code fences.
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
|||||||
13
templates/Prompt/Subblock-Crossblock.md
Normal file
13
templates/Prompt/Subblock-Crossblock.md
Normal file
@@ -0,0 +1,13 @@
|
|||||||
|
Das Lernbaustein-Inventar zum Thema "{topic}" ist in Blöcke mit je eigenen Subbausteinen zerlegt. Manche Aussage taucht in ZWEI Blöcken auf — im Lernguide steht sie dann doppelt. Für jedes Paar unten: Treffen A und B DIESELBE Aussage?
|
||||||
|
|
||||||
|
PAARE (jeweils mit Block-Zugehörigkeit und Kernpunkten):
|
||||||
|
{pairs}
|
||||||
|
|
||||||
|
## Entscheidung pro Paar
|
||||||
|
- **DIESELBE Aussage** (auch anders formuliert, oder eine ist Teilmenge der anderen): Welcher Block ist die natürliche Heimat der Aussage? → antworte **"a"** (A behält sie, B verliert sie) oder **"b"** (B behält sie, A verliert sie). Heimat ist der Block, in dessen Kernthema die Aussage gehört — nicht der, der sie nur am Rand streift.
|
||||||
|
- **VERSCHIEDENE Aussagen** → antworte **"nein"**. Dazu zählt: gleiche Regel, aber auf VERSCHIEDENE Kontexte angewendet (die Anwendung im jeweiligen Block-Kontext ist eigener Lernstoff); Grundregel vs. Sonderfall; gegensätzliche Aussagen.
|
||||||
|
- Im Zweifel: **"nein"**.
|
||||||
|
|
||||||
|
Antworte NUR mit JSON, ohne weiteren Text (jede Paar-Nummer mit "a", "b" oder "nein"):
|
||||||
|
{{"pairs": {{"1": "a", "2": "nein"}}}}
|
||||||
|
{extra}
|
||||||
29
templates/Prompt/Subblock-Konsolidierung.md
Normal file
29
templates/Prompt/Subblock-Konsolidierung.md
Normal file
@@ -0,0 +1,29 @@
|
|||||||
|
Du prüfst die Zerlegung EINES Lernbausteins in Subbausteine für das Thema "{topic}". Jeder Subbaustein wird später ein eigener Absatz im Lernguide. Ziel ist eine 100%-Zerlegung: keinen Eintrag kann man weglassen, ohne dass eine Lücke entsteht — und keinen ergänzen, ohne dass Dopplung entsteht.
|
||||||
|
|
||||||
|
BLOCK: {block}
|
||||||
|
|
||||||
|
SUBBAUSTEINE (nummeriert, mit ihren Kernpunkten):
|
||||||
|
{subs}
|
||||||
|
|
||||||
|
## Auftrag (vier Urteile)
|
||||||
|
1. **gruppen — Dopplungen**: Einträge, die DIESELBE Aussage treffen (anders formuliert) oder reine TEILMENGE eines anderen sind. Pro Gruppe: `haupt` = der GRUNDLEGENDSTE, allgemeinste Eintrag (Basis vor Detail, Grundsyntax vor Sonderfall, nie ein inhaltsleerer Stub).
|
||||||
|
2. **kataloge — Aufzählungs-Bündel**: Einträge, die REINE Nachschlage-Listen gleicher Art sind (z. B. Symbol-, Befehls-, Namenslisten ohne eigene Verhaltensregel). Solche Listen bündelst du zu EINEM Katalog-Eintrag mit sprechendem Titel. Einträge mit eigener Regel oder eigenem Verhalten gehören NICHT in ein Bündel.
|
||||||
|
3. **fremd — Weglass-Test**: Einträge, deren Aussage nicht zum Thema "{topic}" gehört (anderes Fachgebiet, Grundlagen einer anderen Technologie) — auch wenn die Aussage stimmt und belegbar ist. Ließe man sie weg, fehlte dem Thema nichts.
|
||||||
|
4. **luecken**: Ein KERN-Aspekt des Blocks fehlt — ohne ihn ist die Zerlegung UNVOLLSTÄNDIG (die Grundregel oder ein Pflichtbestandteil fehlt). Katalog-, Detail- und Randwissen ist KEINE Lücke. Was man ergänzen KÖNNTE, ist keine Lücke — nur was fehlen DARF nicht. Im Zweifel: keine Lücke. Meist ist die Liste leer.
|
||||||
|
|
||||||
|
## Regeln — der häufigste Fehler zuerst
|
||||||
|
- **Verschiedene Schreibweisen, Marker, Befehle oder Syntaxen sind IMMER eigene Einheiten** — auch wenn sie derselben Kategorie angehören oder im selben Satz erklärt werden könnten. „Gehört zusammen" oder „ähnliches Thema" ist KEIN Gruppierungs-Grund. Nur „sagt DASSELBE" zählt.
|
||||||
|
- Zwei Einträge zum gleichen Konzept mit VERSCHIEDENEN Facetten (Grundregel vs. Sonderfall vs. Randbedingung) sind KEINE Dopplung.
|
||||||
|
- Eine benannte Variante oder ein Spezialfall ist NIE Dopplung seiner Basis.
|
||||||
|
- Einträge mit gegensätzlicher Aussage (Negation) nie gruppieren.
|
||||||
|
- Ein inhaltsleerer oder generischer Eintrag, dessen Titel ein anderer Eintrag vollständig abdeckt, ist Teilmenge → gruppieren, der konkrete Eintrag ist `haupt`.
|
||||||
|
- Im Zweifel: NICHT gruppieren, NICHT bündeln, NICHT als fremd markieren.
|
||||||
|
|
||||||
|
Antworte NUR mit JSON, ohne weiteren Text:
|
||||||
|
{{"gruppen": [{{"haupt": 1, "weitere": [4]}}], "kataloge": [{{"titel": "Symbolkatalog: Operatoren und Relationen", "mitglieder": [2, 5, 9]}}], "fremd": [7], "luecken": ["fehlender Kernaspekt"]}}
|
||||||
|
|
||||||
|
- gruppen: pro Gruppe haupt + weitere (Nummern). Keine Dopplungen → [].
|
||||||
|
- kataloge: pro Bündel Titel + Mitglieds-Nummern. Keine → [].
|
||||||
|
- fremd: Nummern themenfremder Einträge. Keine → [].
|
||||||
|
- luecken: fehlende Kernaspekte in Stichworten. Keine → [].
|
||||||
|
{extra}
|
||||||
13
templates/QA/QA-Bausteine.md
Normal file
13
templates/QA/QA-Bausteine.md
Normal file
@@ -0,0 +1,13 @@
|
|||||||
|
You are an INDEPENDENT quality auditor for a learning-block inventory on the topic "{topic}". For each block title below decide: is this a GENUINE standalone learning block — one self-contained concept/method/theorem a learner can study → **ja** — or is it NOT a real block → **nein**?
|
||||||
|
|
||||||
|
**nein** applies to: bare notation or symbol fragments ("αÜbergang", "|x|"), truncated/mangled titles, exercise references, meta/organizational entries, or titles so vague no one could say what to learn.
|
||||||
|
|
||||||
|
BLOCKS:
|
||||||
|
{blocks}
|
||||||
|
|
||||||
|
When in doubt about the SCOPE but the concept is real → ja. When the title is not a learnable concept → nein.
|
||||||
|
|
||||||
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
Format (one verdict per number):
|
||||||
|
{{"relevant": {{"1": "ja", "2": "nein"}}}}
|
||||||
|
{extra}
|
||||||
13
templates/QA/QA-Dubletten.md
Normal file
13
templates/QA/QA-Dubletten.md
Normal file
@@ -0,0 +1,13 @@
|
|||||||
|
You are an INDEPENDENT quality auditor for a learning-block inventory on the topic "{topic}". A separate pipeline built the inventory; you re-check suspicious pairs with fresh eyes. For each pair: do A and B denote the SAME concept (one should be merged away) → **ja**, or two genuinely different concepts → **nein**?
|
||||||
|
|
||||||
|
SUSPICIOUS PAIRS:
|
||||||
|
{pairs}
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Same concept under two names/spellings/facets → ja. A named variant, a relation, or a different scope → nein.
|
||||||
|
- Judge on the meaning, not the wording. When genuinely unsure → nein.
|
||||||
|
|
||||||
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
Format (one verdict per number):
|
||||||
|
{{"relevant": {{"1": "ja", "2": "nein"}}}}
|
||||||
|
{extra}
|
||||||
15
templates/QA/QA-Guide-Fakten.md
Normal file
15
templates/QA/QA-Guide-Fakten.md
Normal file
@@ -0,0 +1,15 @@
|
|||||||
|
You are an INDEPENDENT quality auditor for a learning guide on the topic "{topic}". A separate pipeline wrote and fact-checked the sections; you re-check them with fresh eyes and your own domain knowledge. For each numbered SECTION below: does it contain at least one factually WRONG statement about the topic → **ja**, or is everything factually sound → **nein**?
|
||||||
|
|
||||||
|
SECTIONS (excerpts):
|
||||||
|
{sections}
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Judge factual CORRECTNESS only — not style, length, completeness or didactics.
|
||||||
|
- Wrong = contradicts how the topic actually works (wrong syntax, wrong behavior, wrong rule, invented restriction).
|
||||||
|
- Simplifications that are broadly correct for a beginner are NOT wrong.
|
||||||
|
- When genuinely unsure → nein.
|
||||||
|
|
||||||
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
Format (one verdict per number):
|
||||||
|
{{"relevant": {{"1": "nein", "2": "ja"}}}}
|
||||||
|
{extra}
|
||||||
13
templates/QA/QA-Luecken.md
Normal file
13
templates/QA/QA-Luecken.md
Normal file
@@ -0,0 +1,13 @@
|
|||||||
|
You are an INDEPENDENT quality auditor for a learning-block inventory on the topic "{topic}". Below are source-material sections that the inventory maps to NO block. For each section: does it contain exam-relevant substance that DESERVES its own block (a definition, method, theorem, technique the inventory misses) → **ja**, or is it filler/organizational/already covered implicitly → **nein**?
|
||||||
|
|
||||||
|
UNCOVERED SECTIONS:
|
||||||
|
{sections}
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Organizational text, prefaces, repeated examples of covered concepts → nein.
|
||||||
|
- A concrete concept the material teaches but no block captures → ja.
|
||||||
|
|
||||||
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
Format (one verdict per number):
|
||||||
|
{{"relevant": {{"1": "ja", "2": "nein"}}}}
|
||||||
|
{extra}
|
||||||
14
templates/QA/QA-Repair-Beleg.md
Normal file
14
templates/QA/QA-Repair-Beleg.md
Normal file
@@ -0,0 +1,14 @@
|
|||||||
|
You are re-checking EXISTING inventory blocks on the topic "{topic}" that a QA scan flagged as possibly NOT covered by the learning material. For each block: do the excerpts show the material actually treats this block → **ja** (keep), or is there truly no coverage → **nein** (remove)?
|
||||||
|
|
||||||
|
FLAGGED BLOCKS (each with its best-matching material excerpts):
|
||||||
|
{blocks}
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- **ja** = any real coverage: a heading, definition, algorithm, proof, exercise or substantial passage about the block's subject — also under a synonym, symbol variant or hyphenated spelling (PDF extraction mangles titles).
|
||||||
|
- **nein** ONLY when the excerpts clearly have nothing to do with the block.
|
||||||
|
- This removes existing content, so when in doubt → **ja**.
|
||||||
|
|
||||||
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
Format (one verdict per number):
|
||||||
|
{{"relevant": {{"1": "ja", "2": "nein"}}}}
|
||||||
|
{extra}
|
||||||
17
templates/QA/QA-Sub-Dubletten.md
Normal file
17
templates/QA/QA-Sub-Dubletten.md
Normal file
@@ -0,0 +1,17 @@
|
|||||||
|
You are an INDEPENDENT quality auditor for a learning-block inventory on the topic "{topic}". Each block is decomposed into sub-blocks; every sub-block becomes its own paragraph in the learning guide. A separate pipeline built them; you re-check suspicious pairs with fresh eyes. For each pair: do A and B state the SAME thing (one would be redundant in the guide) → **ja**, or two genuinely different statements → **nein**?
|
||||||
|
|
||||||
|
Each entry names its block in [brackets] — the pair may span two different blocks.
|
||||||
|
|
||||||
|
SUSPICIOUS PAIRS:
|
||||||
|
{pairs}
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
- Same statement in different wording, or one being a pure subset of the other → ja.
|
||||||
|
- Same concept but DIFFERENT facets (base rule vs. special case vs. edge condition) → nein.
|
||||||
|
- The same rule applied to two DIFFERENT contexts/blocks → nein (each application is its own learning content).
|
||||||
|
- Opposite statements (negation) → nein. When genuinely unsure → nein.
|
||||||
|
|
||||||
|
Reply with ONLY the JSON — no other text, no code fences.
|
||||||
|
Format (one verdict per number):
|
||||||
|
{{"relevant": {{"1": "ja", "2": "nein"}}}}
|
||||||
|
{extra}
|
||||||
Reference in New Issue
Block a user