"""Blocks pipeline: research consensus + clarification loop — pure inventory, unsorted. 5x research (min. 3, grace) → mapping (consensus/rest) → clarification loop (max. CONSENSUS_MAX_ROUNDS rounds): 3 selection agents (min. 2, grace) decide on the disputed rest, a mapping agent sorts into accept/discard/ still disputed. An empty rest ends the loop; the last round must decide everything. Races use a grace window instead of "first N win": after the first valid result, the remaining agents get CONSENSUS_GRACE seconds to finish. The consensus is accumulated in code — no agent re-emits the full list. """ import asyncio import hashlib import json import logging import math import re import shutil import subprocess import unicodedata from pathlib import Path import database as db import embedding from agents import kill_process, cancel_scope, clear_scope from config import CONSENSUS_GRACE, CONSENSUS_MAX_ROUNDS, DEFAULT_PROVIDER, EMBEDDING_AKTIV, EMBEDDING_SUB_DUP, SUB_VARIANT_COS, SUBBLOCK_MAX, EVIDENCE_BUDGET_CHARS, EVIDENCE_CTX_LINES from fsutil import atomic_write_json from jsonio import parse_json_text, read_json_file as _json_file from paths import arbeit_dir, blocks_path, question_pattern_path, project_dir, subblocks_path, source_path, source_crawl_dir, safe_folder from crawl import load_pages from pipeline import ( GenContext, _extra, _gather_progress, _yesno_schema, _log, _prompt, _race, _semaphore, _timeout, run_single_slot, ) from textkit import _load_blocks, _norm_title, _parse_selection, _title # Pipeline-Tuning-Konstanten liegen zentral in config.py (tunebar via CREATOR_PARAMS). from config import RESEARCH_SECTION_CHARS # noqa: E402 log = logging.getLogger("creator.blocks") _blocks_progress: dict[str, str] = {} _blocks_errors: dict[str, str] = {} _blocks_cancelled: set[str] = set() _blocks_step: dict[str, int] = {} ARTEFACT_TYPES = ("flashcard", "example") def load_source(topic: str) -> dict: """Read the persisted source choice. Fallback (legacy topics without source.json): if projects/ exists → projekt, otherwise thema.""" q = _json_file(source_path(topic)) if isinstance(q, dict) and q.get("type") in ("thema", "projekt", "uni", "link"): return q if project_dir(topic).is_dir(): return {"type": "projekt", "location": f"projects/{topic}", "spec": ""} return {"type": "thema", "location": "", "spec": ""} def source_folder(topic: str) -> Path | None: """Folder source (projekt/uni → path, link → crawl folder) — otherwise None (thema).""" q = load_source(topic) if q["type"] == "link": return source_crawl_dir(topic) if q["type"] in ("projekt", "uni"): return safe_folder(q.get("location", "")) return None def _crawl_done(topic: str) -> bool: return (source_crawl_dir(topic) / ".done").exists() # marker only on clean completion # Learning-path levels (beginner/advanced/expert); old difficulty values are backward-compatible. _LEVELS = ("beginner", "advanced", "expert", "easy", "medium", "hard") async def subblocks_title(topic: str, block: str) -> list[str]: """Subblock titles of a block — DB-first (consensus), fallback to the sidecar file.""" rows = [s["sub_title"] for s in await db.list_subblocks(topic, _norm_title(block)) if s["status"] == "consensus" and s["sub_title"]] if rows: return rows sc = _json_file(subblocks_path(topic)) if not isinstance(sc, dict): return [] return [ t for s in (sc.get(block) or []) if isinstance(s, dict) and (t := str(s.get("title", "")).strip()) ] async def load_question_pattern(topic: str, block: str) -> list[dict]: """Predefined question patterns of a block — DB-first, fallback to sidecar (empty = live).""" rows = await db.list_question_pattern(topic, _norm_title(block)) if rows: return [{"subblock": r["sub_title"], "question": r["question"]} for r in rows if r["question"]] fm = _json_file(question_pattern_path(topic)) if not isinstance(fm, dict): return [] return [ {"subblock": str(e.get("subblock", "")).strip(), "question": question} for e in (fm.get(block) or []) if isinstance(e, dict) and (question := str(e.get("question", "")).strip()) ] async def subblocks_frei(topic: str, block: str, max_level: int) -> list[str]: """Subblock titles up to the unlocked level (≤ max_level). Fallback without level knowledge (legacy/sidecar): all subblock titles.""" rows = await db.subs_with_level(topic, block) if not rows: return await subblocks_title(topic, block) return [s["title"] for s in rows if s["level"] <= max_level and s["title"]] async def load_question_pattern_free(topic: str, block: str, max_level: int) -> list[dict]: """Question patterns, filtered to subblocks up to the unlocked level. Without level knowledge (legacy/sidecar), unfiltered.""" rows = await db.subs_with_level(topic, block) if not rows: return await load_question_pattern(topic, block) unlocked = {s["norm"] for s in rows if s["level"] <= max_level} return [m for m in await load_question_pattern(topic, block) if _norm_title(m["subblock"]) in unlocked] async def load_overview(topic: str) -> list[dict]: """Structured block list for the overview — DB-first (consensus + subs/levels/relevance), fallback to blocks.md + sidecar (legacy topics).""" bs = await db.list_blocks(topic, status="consensus") if bs: out = [] for num, b in enumerate(bs, 1): subs = [s for s in await db.list_subblocks(topic, b["title_norm"]) if s["status"] == "consensus"] out.append({ "num": num, "title": b["title"], "description": b["description"], "subblocks": [ {"title": s["sub_title"], "level": s["level"] if s["level"] in _LEVELS else "advanced", "relevance": s["relevance"] if s["relevance"] in ("relevant", "peripheral") else None} for s in subs if s["sub_title"] ], }) return out entries = _load_blocks(_read(blocks_path(topic))) sidecar = _json_file(subblocks_path(topic)) sidecar = sidecar if isinstance(sidecar, dict) else {} out = [] for num, entry in entries.items(): title = _title(entry) split_parts = entry.split(" — ", 1) description = split_parts[1].strip() if len(split_parts) == 2 else "" subblocks = [ { "title": t, "level": s.get("level") if s.get("level") in _LEVELS else "advanced", "relevance": s.get("relevance") if s.get("relevance") in ("relevant", "peripheral") else None, } for s in (sidecar.get(title) or []) if isinstance(s, dict) and (t := str(s.get("title", "")).strip()) ] out.append({"num": num, "title": title, "description": description, "subblocks": subblocks}) return out def _blocks_steps(topic: str) -> tuple: """Steps per source: link gets "Source laden" up front, projekt additionally "Supplement". Subblocks + levels are three phases each (find, select, clarify). Per phase all packages run in parallel; the step remains until the last package is done. """ q = load_source(topic) base = ("Research", "Consolidation", "Clarification", "Dedup", "Blocks-Filter", "Blocks-Gruppierung") rest = ( "Generate", "Verify", "Fix", "Outline", "Artefakte gen", "Artefakte check", ) middle = base + (("Supplement",) if q["type"] == "projekt" else ()) + rest return (("Source prep",) if q["type"] == "link" else ()) + middle def _step_idx(topic: str, name: str) -> int: return _blocks_steps(topic).index(name) def _report_p(set_p, topic: str, step: str): """Async report callback for _gather_progress: sets " d/t…" + step index.""" idx = _step_idx(topic, step) async def report(d, t): set_p(f"{step} {d}/{t}…", step=idx) return report # Coarse display phases: bundle the fine steps (internally everything stays fine-grained). # Special steps (Source laden, Supplement) belong to the "Inventory" phase. PHASEN = ( ("Source", ("Source prep",)), ("Inventory", ("Research", "Consolidation", "Clarification", "Dedup", "Blocks-Filter", "Blocks-Gruppierung", "Supplement")), ("Generate", ("Generate",)), ("Verify", ("Verify", "Fix")), ("Outline", ("Outline",)), ("Artefakte", ("Artefakte gen", "Artefakte check")), ) def _blocks_files(topic: str) -> dict: work_dir = arbeit_dir(topic) rounds = range(1, CONSENSUS_MAX_ROUNDS + 1) return { "final": blocks_path(topic), "arbeit": work_dir, "research": [work_dir / f"research-{i}.md" for i in (1, 2, 3, 4, 5)], "research_mapping": work_dir / "research-mapping.json", "selection": {n: [work_dir / f"selection-r{n}-{i}.json" for i in (1, 2, 3)] for n in rounds}, "mapping": {n: work_dir / f"selection-mapping-r{n}.json" for n in rounds}, "ergaenzung": work_dir / "ergaenzung.json", "sub_roh": work_dir / "subblocks-roh.json", "facts": work_dir / "subblocks-facts.json", "sidecar": subblocks_path(topic), "question_pattern": question_pattern_path(topic), "outline": work_dir / "outline.json", "outline_slots": [work_dir / f"outline-{i}.json" for i in (1, 2, 3)], "artefakte": work_dir / "artefakte.json", } def cancel_blocks(topic: str) -> bool: if topic not in _blocks_progress: return False _blocks_cancelled.add(topic) cancel_scope(f"blocks-{topic}-") # waiting agents bail before spawning kill_process(f"blocks-{topic}-") # kill running subprocesses return True async def blocks_status(topic: str) -> dict: """Kanban-based status: `generating` from the run registry, progress from the card counts. `partial` = cards sit in non-terminal columns while nothing runs (continue-able).""" generating = topic in _blocks_progress counts = await db.kanban_stage_counts(topic) # ready = finished inventory. The DB is the source of truth (a synced topic may lack # blocks.md, the file is a legacy mirror) — either signal counts. ready = blocks_path(topic).exists() or counts.get("inventory", {}).get("done_block", 0) > 0 terminal = {"clustered", "done_cluster", "grouped", "rejected", "done_block", "done_artefact", "dead"} open_cards = sum(n for stages in counts.values() for stage, n in stages.items() if stage not in terminal) return { "ready": ready, "generating": generating, "progress": _blocks_progress.get(topic), "error": _blocks_errors.get(topic), "partial": not generating and open_cards > 0, } def active_blocks() -> list[dict]: return [{"topic": t, "progress": p} for t, p in _blocks_progress.items()] def reset_blocks(topic: str) -> None: """"Remove": deletes the ENTIRE blocks area — crawl, triage, inventory … questions. KEEPS only the topic config `source.json` (type/link/spec). Re-generating crawls anew. (Crawl/triage belong to the blocks; only the config is the "topic".)""" files = _blocks_files(topic) files["final"].unlink(missing_ok=True) files["sidecar"].unlink(missing_ok=True) files["question_pattern"].unlink(missing_ok=True) shutil.rmtree(source_crawl_dir(topic), ignore_errors=True) # crawl belongs to the blocks shutil.rmtree(files["arbeit"], ignore_errors=True) _blocks_errors.pop(topic, None) # source.json intentionally stays — that is the topic config. def _supplement_schema(data): """{"blocks": [{"title", "description"}]} → list (empty allowed) · otherwise None.""" if not isinstance(data, dict) or not isinstance(data.get("blocks"), list): return None out = [] for b in data["blocks"]: if not isinstance(b, dict) or not isinstance(b.get("title"), str) or not isinstance(b.get("description"), str): return None title, description = b["title"].strip(), b["description"].strip() if not title: return None out.append((title, description)) return out def _ocr_languages() -> str | None: """Installierte Tesseract-Sprachen (deu/eng), None → OCR aus.""" try: import pymupdf base = Path(pymupdf.get_tessdata()) except Exception: return None langs = [l for l in ("deu", "eng") if (base / f"{l}.traineddata").exists()] return "+".join(langs) or None def _pdf_markdown(pdf: Path) -> str | None: """pymupdf4llm → Markdown string (None if the lib is missing or it fails).""" try: import pymupdf4llm except ImportError: return None try: # OCR nur, wenn Tesseract-Sprachdaten wirklich vorhanden sind — # sonst wirft der OCR-Pfad und die ganze Datei faellt auf pdftotext. langs = _ocr_languages() kwargs = {"use_ocr": True, "ocr_language": langs} if langs else {"use_ocr": False} return pymupdf4llm.to_markdown(str(pdf), show_progress=False, **kwargs) except Exception: log.warning("pymupdf4llm failed for %s", pdf.name, exc_info=True) return None def _pdf_plaintext(pdf: Path) -> str | None: """pdftotext -layout → plain string (None if missing/fails).""" if shutil.which("pdftotext") is None: return None try: out = subprocess.run(["pdftotext", "-layout", str(pdf), "-"], check=True, timeout=120, capture_output=True) return out.stdout.decode("utf-8", errors="replace") except Exception: log.warning("pdftotext failed for %s", pdf.name, exc_info=True) return None # Content-fidelity guard between the two converters (topic-NEUTRAL: measures loss, not domain). # pymupdf4llm yields structured Markdown but silently DROPS rendered display formulas and can # splinter combining diacritics (measured on a LaTeX script: „P = {L …}" gone, „h¨aufig"). # pdftotext is structure-poor but faithful. Take the Markdown only when it preserves the bulk # of the content; otherwise the faithful plaintext wins. _PDF_FIDELITY_SYMBOLS = "≤≥∈∉⊆∪∧∨¬→Σδα{}=" def _pick_conversion(md: str | None, plain: str | None) -> tuple[str, str] | None: if md is None and plain is None: return None if md is None: return plain, "pdftotext" if plain is None: return md, "pymupdf4llm" ok_len = len(md) >= 0.7 * len(plain) sym_plain = sum(plain.count(c) for c in _PDF_FIDELITY_SYMBOLS) ok_sym = sym_plain == 0 or sum(md.count(c) for c in _PDF_FIDELITY_SYMBOLS) >= 0.8 * sym_plain ok_diakritik = md.count("\u00a8") <= plain.count("\u00a8") + 2 # standalone ¨ = splintered umlauts if ok_len and ok_sym and ok_diakritik: return md, "pymupdf4llm" return plain, "pdftotext" def _convert_pdfs(project: Path) -> None: """Convert PDFs in the project to .txt — agents read text instead of page images. Called before every project generation; converts only if the .txt is missing or older than the PDF. Both converters run; the fidelity guard picks the better result per file. Neither available → hard error instead of an unreliable direct-read mode (MiniMax image limit, vision cost).""" pdfs = list(project.rglob("*.pdf")) if not pdfs: return for pdf in pdfs: txt = pdf.with_suffix(".txt") if txt.exists() and txt.stat().st_mtime >= pdf.stat().st_mtime: continue picked = _pick_conversion(_pdf_markdown(pdf), _pdf_plaintext(pdf)) if picked is None: raise RuntimeError(f"PDF conversion failed ({pdf.name}): weder pymupdf4llm noch " "pdftotext verfügbar/erfolgreich (pip install pymupdf4llm oder poppler-utils)") text, tool = picked txt.write_text(text, encoding="utf-8") _log(project.name, f"PDF konvertiert ({tool}): {pdf.name} → {txt.name}") _SOURCE_TEMPLATE = {"projekt": "Blocks-Source-Projekt", "uni": "Blocks-Source-Uni", "link": "Blocks-Source-Link"} def _text_sections(text: str, goal: int = RESEARCH_SECTION_CHARS) -> list[str]: """Split text at paragraph/line boundaries into sections of ~`ziel` chars (against lost-in-the-middle on large documents). Small text stays ONE section. Content stays complete — only separating whitespace is dropped.""" text = text.strip() if len(text) <= goal: return [text] if text else [] sections: list[str] = [] buf = "" def flush(): nonlocal buf if buf.strip(): sections.append(buf.strip()) buf = "" for block in re.split(r"\n\s*\n", text): # at paragraph boundaries block = block.strip() if not block: continue if len(block) > goal: # single huge paragraph → hard-cut at lines flush() for line in block.split("\n"): if buf and len(buf) + len(line) + 1 > goal: flush() buf += line + "\n" flush() elif buf and len(buf) + len(block) + 2 > goal: flush() buf = block else: buf = (buf + "\n\n" + block) if buf else block flush() return sections # ── Inline evidence for judges ────────────────────────────────────────────────────── # Judges used to re-search the corpus per session ({source} → "ls/find … read", ~10 tool # turns each). The corpus excerpts now go INTO the prompt; the agent answers as text. def _corpus_files(folder: Path | None, sources: list[str] | None) -> list[Path]: """The block's named source .txt files; fallback: every .txt in the folder.""" if folder is None or not folder.is_dir(): return [] if sources: named = [folder / Path(s).with_suffix(".txt").name for s in sources if s] named = [p for p in named if p.is_file()] if named: return named return sorted(p for p in folder.glob("*.txt") if p.is_file()) def _q_tokens(text: str) -> set[str]: return set(re.findall(r"\w{3,}", text.casefold())) def _evidence_pack(folder: Path | None, sources: list[str] | None, queries: list[str], budget: int = EVIDENCE_BUDGET_CHARS) -> str: """Keyword-selected corpus excerpts for a judge prompt. Sections are ranked by token overlap with `queries` (block title + candidates); every query with any match gets its best section (coverage guarantee), the rest of the budget takes the global top. Empty string when there is no corpus — the caller keeps the old self-research source then.""" parts: list[tuple[str, int, str, set[str]]] = [] # (file, idx, text, tokens) for f in _corpus_files(folder, sources): try: text = f.read_text(encoding="utf-8") except OSError: continue for i, sec in enumerate(_text_sections(text), 1): parts.append((f.name, i, sec, _q_tokens(sec))) if not parts: return "" qtoks = [(_q_tokens(q)) for q in queries if q] score = [sum(len(qt & p[3]) for qt in qtoks) for p in parts] chosen: set[int] = set() for qt in qtoks: # coverage guarantee: best section per query best = max(range(len(parts)), key=lambda k: len(qt & parts[k][3]), default=None) if best is not None and qt & parts[best][3]: chosen.add(best) used = sum(len(parts[k][2]) for k in chosen) for k in sorted(range(len(parts)), key=lambda k: -score[k]): # top-up to budget if k in chosen or score[k] <= 0: continue if used + len(parts[k][2]) > budget: continue chosen.add(k) used += len(parts[k][2]) out, total = [], 0 for k in sorted(chosen): # document order for readability fname, i, sec, _t = parts[k] if total + len(sec) > max(budget, used): # hard cap incl. guarantee overshoot break out.append(f"── {fname} · Abschnitt {i} ──\n{sec}") total += len(sec) return "\n\n".join(out) _CITE_POS = re.compile(r"\b(?:Z(?:eilen?)?|lines?)\.?\s*(\d+)(?:\s*[-–]\s*(\d+))?", re.I) def _cite_ref(cite: str, files: list[Path]) -> tuple[Path, int, int] | None: """(file, line_lo, line_hi) from a cited_facts source string („Skript.txt, Übung 6.47, Z.1341-1344") — None when file or position is missing.""" c = (cite or "").casefold() f = next((p for p in files if p.name.casefold() in c or p.stem.casefold() in c), None) m = _CITE_POS.search(cite or "") if f is None or m is None: return None lo, hi = int(m.group(1)), int(m.group(2) or m.group(1)) return (f, min(lo, hi), max(lo, hi)) def _cited_evidence(folder: Path | None, sources: list[str] | None, cites: list[str], fallback_queries: list[str], budget: int = EVIDENCE_BUDGET_CHARS) -> str: """Evidence for the facts check: the EXACT cited regions (±EVIDENCE_CTX_LINES, merged, line numbers in the header) — precise and tiny. Cites without a parseable position fall back to the keyword pack. Empty string without a corpus.""" files = _corpus_files(folder, sources) if not files: return "" ranges: dict[Path, list[tuple[int, int]]] = {} unresolved = False for c in cites: ref = _cite_ref(c, files) if ref is None: unresolved = True continue f, lo, hi = ref ranges.setdefault(f, []).append((max(1, lo - EVIDENCE_CTX_LINES), hi + EVIDENCE_CTX_LINES)) out, total = [], 0 for f in files: if f not in ranges: continue try: lines = f.read_text(encoding="utf-8").splitlines() except OSError: continue merged: list[list[int]] = [] for lo, hi in sorted(ranges[f]): hi = min(hi, len(lines)) if merged and lo <= merged[-1][1] + 1: merged[-1][1] = max(merged[-1][1], hi) else: merged.append([lo, hi]) for lo, hi in merged: sec = "\n".join(lines[lo - 1:hi]) if not sec.strip() or total + len(sec) > budget: continue out.append(f"── {f.name} · Z. {lo}-{hi} ──\n{sec}") total += len(sec) if unresolved or not out: pack = _evidence_pack(folder, sources, fallback_queries, max(0, budget - total)) if pack: out.append(pack) return "\n\n".join(out) def _reply_text(result) -> str: """Assistant text of a no-tool agent call ((rc, stdout, stderr) from run_agent).""" return (result[1] or "") if result else "" def _sink_or_file(result, path: Path, schema): """Text reply preferred, file fallback. Research agents used to WRITE their big JSON — measured: one facts call spent 40 of 64 turns in a write/validate/repair loop (9 min). As text, parse_json_text repairs the escaping in one pass; tools stay for research.""" val = _sink_json(result, path, schema) if val is not None: return val return schema(_json_file(path)) def _sink_json(result, path: Path, schema): """Payload validator for no-tool agents: the JSON comes as reply TEXT; the engine persists it to `path`, so resume guards and audit files keep working unchanged.""" data = parse_json_text(_reply_text(result)) val = schema(data) if val is not None: atomic_write_json(path, data) return val def material_folder(topic: str) -> Path | None: """Korpus fürs Inline-Material: echte Quelle (uni/projekt/link) oder bei thema die Fundstellen der Research-Reader (arbeit/material/*.txt). Nur für Evidence-Packs — Konsens-Gate und QA messen unverändert gegen die konfigurierte Quelle (Messinvarianz).""" f = source_folder(topic) if f is not None: return f d = arbeit_dir(topic) / "material" return d if d.is_dir() and any(d.glob("*.txt")) else None def _build_research_prompt(topic: str, out_path: Path, instructions: str, type: str, folder: Path | None, fokus: str = "", section: str = "", material_path: Path | None = None) -> str: if section: # Section mode (uni/projekt): text directly in the prompt → small context, no file reading. source = section elif type in _SOURCE_TEMPLATE: source = _prompt(_SOURCE_TEMPLATE[type], project=folder) else: source = _prompt("Blocks-Source-Thema", topic=topic) if material_path is not None: # thema: Fundstellen sichern — sie werden das Inline-Material der Folgeschritte # (Finder/Facts liefen sonst mit eigener Websuche → Reasoning-Schleifen, Retries) source += ("\n\nALSO write the file " + str(material_path) + " as you research: for every " "source you use, append one line with the URL followed by the relevant excerpt " "(plain text). Later steps back all facts ONLY against this material — " "an excerpt you skip here cannot back anything later.") return _prompt( "Blocks-Research", topic=topic, source=source, blocks_path=out_path, focus=fokus, extra=_extra(instructions), ) def _file_payload(path: Path): """Valid if the slot file exists and contains numbered entries.""" if not path.exists(): return None text = path.read_text(encoding="utf-8") return text if _parse_selection(text) else None def _read(p: Path) -> str: return p.read_text(encoding="utf-8") if p.exists() else "" def _chunk_nums(items: list, n: int) -> list[list]: """Splits a flat list into n chunks as equal in size as possible.""" n = max(1, n) size = max(1, math.ceil(len(items) / n)) return [items[i:i + size] for i in range(0, len(items), size)] def _n_chunks(count: int, size: int) -> int: return min(SUBBLOCK_MAX, max(1, math.ceil(count / size))) _NEG_LEMMA = {"nicht": "nicht", "ohne": "ohne", "nie": "nie", "niemals": "nie", "kein": "kein", "keine": "kein", "keinen": "kein", "keiner": "kein", "keinem": "kein", "keines": "kein"} def _neg_set(title: str) -> frozenset: """Lemmatized negation tokens of a title — antonym statements measure cos 0.91–0.95 (above any usable variant threshold), so equal negation sets are a hard merge precondition.""" return frozenset(l for t in re.findall(r"\w+", _norm_title(title)) if (l := _NEG_LEMMA.get(t))) def _sub_tokens(title: str) -> set: return set(re.findall(r"\w+", _norm_title(title))) def _variant_clusters(titles: list[str], mentions: list[int], sims) -> list[dict]: """Fold phrasing VARIANTS of one concept BEFORE the consensus count: finders rephrase per round, so exact-norm counting starves real concepts. Union-find over cos ≥ SUB_VARIANT_COS with the negation guard. → [{"rep": idx, "members": [idx…], "mentions": sum}].""" n = len(titles) parent = list(range(n)) def find(x): while parent[x] != x: parent[x] = parent[parent[x]] x = parent[x] return x negs = [_neg_set(t) for t in titles] for i in range(n): for j in range(i + 1, n): if float(sims[i][j]) >= SUB_VARIANT_COS and negs[i] == negs[j]: parent[find(i)] = find(j) groups: dict[int, list[int]] = {} for i in range(n): groups.setdefault(find(i), []).append(i) return [{"rep": max(g, key=lambda k: (len(titles[k]), -k)), "members": g, "mentions": sum(mentions[k] for k in g)} for g in groups.values()] async def _dedup_subblocks(topic: str, raw: dict[str, list[str]]) -> None: """Deterministic near-duplicate filter per block: subblocks with cosine ≥ EMBEDDING_SUB_DUP are the same statement (reliable in the narrow block context — no LLM needed). Per duplicate group keeps the most informative (longest); rest → DB discarded + out of `roh`. Model missing → silently skip (like the rest of the embedding fallback).""" if not EMBEDDING_AKTIV or not await asyncio.to_thread(embedding.available): return for title, subs in list(raw.items()): if len(subs) < 2: continue sims = await asyncio.to_thread(embedding.embed_sims, subs) if sims is None: return keepers: list[int] = [] discarded: list[int] = [] negs = [_neg_set(s) for s in subs] for i in sorted(range(len(subs)), key=lambda x: (-len(subs[x]), x)): # most informative first if any(float(sims[i][j]) >= EMBEDDING_SUB_DUP and negs[i] == negs[j] for j in keepers): discarded.append(i) else: keepers.append(i) if not discarded: continue bnorm = _norm_title(title) for i in discarded: await db.set_subblock_fields(topic, bnorm, _norm_title(subs[i]), status="discarded") raw[title] = [subs[i] for i in sorted(keepers)] # original order of the kept ones def _facts_union(wf: dict, lf: dict) -> None: """Merge a folded sub's facts into the winner's: key_points/cited_facts union (exact-duplicate-free), scalar fields only fill gaps.""" for feld in ("key_points", "cited_facts"): have = wf.get(feld) or [] seen = {json.dumps(e, sort_keys=True, ensure_ascii=False) for e in have} fresh = [e for e in (lf.get(feld) or []) if json.dumps(e, sort_keys=True, ensure_ascii=False) not in seen] if fresh: wf[feld] = have + fresh for feld in ("prerequisites", "hurdles", "example_idea"): if not wf.get(feld) and lf.get(feld): wf[feld] = lf[feld] def _agreed_cliques(pair_sets: list[set], negs: list, n: int) -> list[list[int]]: """Union-find over the UNANIMOUS pairs (both judges grouped them), negation-guarded.""" agreed = {(a, b) for a, b in pair_sets[0] & pair_sets[1] if negs[a - 1] == negs[b - 1]} parent = list(range(n + 1)) def find(x): while parent[x] != x: parent[x] = parent[parent[x]] x = parent[x] return x for a, b in agreed: parent[find(a)] = find(b) groups: dict[int, list[int]] = {} for k in range(1, n + 1): groups.setdefault(find(k), []).append(k) return [g for g in groups.values() if len(g) >= 2] def _pairs_of(groups) -> set: ps: set[tuple[int, int]] = set() for ids in groups: ps |= {(a, b) for x, a in enumerate(ids) for b in ids[x + 1:]} return ps _LUECKEN_CAP = 3 # the gap list feeds ONE finder round — an uncapped list doubled the decomposition def _luecken_schnitt(l1: list[str], l2: list[str], cap: int = _LUECKEN_CAP) -> list[str]: """Gaps BOTH judges name — exact strings never match across paraphrases, so a gap survives when the other judge names one sharing a distinctive token (≥4 chars). j1's phrasing wins. The measured union produced 107 'gaps' on 216 subs.""" def toks(s): return {t for t in _sub_tokens(s) if len(t) >= 4} toks2 = [toks(l) for l in l2] out = [l for l in l1 if toks(l) and any(toks(l) & t2 for t2 in toks2)] return out[:cap] def _match_sub(agent_sub: str, rel: list[str]) -> str: """Map the agent's subblock title to the matching relevant title — exact, then normalized, then substring (the agent drops e.g. the prefix "Question: "). No match → keep the agent title. This way NO pattern is lost to a title mismatch.""" if agent_sub in rel: return agent_sub an = _norm_title(agent_sub) for r in rel: rn = _norm_title(r) if an and rn and (an == rn or an in rn or rn in an): return r return agent_sub def _subs_hash(sidecar_or_raw: dict) -> str: """Sub-set identity for the resume files of the sub-CONSUMING stages (levels/relevance/ questions/artefacts). Without it a re-run with a recut sub set adopted the stale stage results (measured: 626 orphans — artefacts of the old 425-sub set re-imported).""" parts: list[str] = [] for title, subs in sidecar_or_raw.items(): parts.append(str(title)) for s in subs: parts.append(s["title"] if isinstance(s, dict) else str(s)) return hashlib.md5("\n".join(parts).encode()).hexdigest()[:8] _FACTS_FIELDS = ("key_points", "prerequisites", "hurdles", "cited_facts", "example_idea") def _facts_schema(data) -> list[dict] | None: """{"facts": [{block, subblock, …}]} → valid list · otherwise None. Strictly separates belegte_facts (with source) from example_idee (generative).""" if not isinstance(data, dict) or not isinstance(data.get("facts"), list): return None out = [] for e in data["facts"]: if not isinstance(e, dict): continue blk = str(e.get("block", "")).strip() sub = str(e.get("subblock", "")).strip() if not blk or not sub: continue bf = [{"text": t, "source": str(f.get("source", "")).strip()} for f in (e.get("cited_facts") or []) if isinstance(f, dict) and (t := str(f.get("text", "")).strip())] out.append({ "block": blk, "subblock": sub, "key_points": [k for x in (e.get("key_points") or []) if (k := str(x).strip())], "prerequisites": str(e.get("prerequisites", "")).strip(), "hurdles": str(e.get("hurdles", "")).strip(), "cited_facts": bf, "example_idea": str(e.get("example_idea", "")).strip(), }) return out or None def _facts_lines(fk: dict) -> str: z = [] if fk.get("key_points"): z.append("Kernpunkte: " + " · ".join(str(k) for k in fk["key_points"])) if fk.get("prerequisites"): z.append("Voraussetzung: " + fk["prerequisites"]) if fk.get("hurdles"): z.append("Hürde: " + fk["hurdles"]) for bf in fk.get("cited_facts", []): z.append(f"FAKT: {bf['text']} (Source: {bf.get('source', '?')})") if fk.get("example_idea"): z.append("Example: " + fk["example_idea"]) return "\n".join(z) # ── Inventory in the DB: research loop · consolidation · clarification ──────────── def _crawl_index(folder) -> dict[str, str]: """Alias (filename OR QUELLE: URL, lowercase) → canonical page key (filename).""" idx: dict[str, str] = {} if not folder or not Path(folder).is_dir(): return idx for p in sorted(Path(folder).glob("*.txt")): key = p.name idx[key.lower()] = key try: first_line = p.read_text(encoding="utf-8").splitlines()[0] except (OSError, IndexError): first_line = "" if first_line.startswith("QUELLE:"): url = first_line[len("QUELLE:"):].strip() if url: idx[url.lower()] = key idx[url.rstrip("/").lower()] = key return idx async def _prepare_source(ctx: GenContext, set_p, files: dict, q: dict, folder, instructions: str) -> bool: """Step "Source prep": load link pages + PDF convert. → True (ok) / False (cancel/error). link: load each user-supplied URL (one page, no following); ALL pages count as content (curated — no triage). thema: nothing. projekt/uni: only PDFs (curated folder).""" topic, is_cancelled = ctx.topic, ctx.is_cancelled if not folder: return True # thema → no source to prepare if q["type"] != "link": await asyncio.to_thread(_convert_pdfs, folder) # projekt/uni: only PDFs, no triage return True if await db.get_step_status(topic, "Source prep") == "done": return True if not _crawl_done(topic): urls = [ln.strip() for ln in q["location"].splitlines() if ln.strip()] set_p("Loading pages…", step=_step_idx(topic, "Source prep")) n = await asyncio.to_thread(load_pages, urls, folder, cancelled=is_cancelled) if is_cancelled(): return False if not n: _blocks_errors[topic] = "Links yielded no content — check the URLs" return False await asyncio.to_thread(_convert_pdfs, folder) pages = sorted(set(_crawl_index(folder).values())) if pages: # Curated links: the user picked them — keep every page, no rule/LLM triage. await db.delete_coverage(topic) await db.mark_content(topic, pages, []) _log(topic, f"Links: {len(pages)} pages as content (curated, no triage)") await db.set_step_status(topic, "Source prep", "done") return True _ASPECT_MARKER = ("∈ np", "∈np", " in np", "np-schwer", "np-vollständig", "verifizierer", "zertifikat", "ndtm", "nicht-determ", "lower bound", "untere schranke", "bzgl", "als sprache", "beweis", "güte", "austausch", # generic bound/limit/runtime property stems (a "…-Grenze"/"…-Schranke"/"…-Laufzeit" # is a property OF a concept, not the concept — MDM survivorship must never pick it as # the representative, so it scores >0 here like the other aspect markers): "grenze", "schranke", "laufzeit") def _aspect_marker(title: str) -> int: """Number of property markers in the title (∈NP, NP-hard, verifier, lower bound …). 0 = generic main concept (the problem itself); >0 = a property of it.""" t = title.casefold() return sum(1 for m in _ASPECT_MARKER if m in t) _REFERENCE_RE = re.compile(r'^(Satz|Lemma|Korollar|Bemerkung|Definition)\s*[\d.]+\s*(\([a-z]\)|[a-z])?\s*$', re.I) # Catalogue scaffolding around a concept name: a leading label+number ("Satz 7.13", "Satz 7.6:") # or a trailing "(Definition 6.19)". _reference_strip peels it off and returns the concept remainder; # _is_reference is then just "nothing survives the strip". Generic — no author/domain whitelist. The # remainder feeds the rename step (a card must be titled by its concept, not its catalogue number). _REF_WORDS = (r'(?:Satz|Lemma|Korollar|Corollary|Bemerkung|Beobachtung|Definition|Def' r'|Theorem|Proposition|Folgerung|Kapitel|Abschnitt)') _REF_PREFIX_RE = re.compile(rf'^\s*{_REF_WORDS}\s*\d+(?:\.\d+)*[a-z]?\s*[:.\-–—]?\s*', re.I) _REF_SUFFIX_RE = re.compile(rf'\s*[\(\[]\s*{_REF_WORDS}\s*\d+(?:\.\d+)*[a-z]?\s*[\)\]]\s*$', re.I) _WHOLE_PAREN_RE = re.compile(r'^\s*[\(\[]\s*(.{2,}?)\s*[\)\]]\s*$') def _reference_strip(title: str) -> str: """Concept remainder of a catalogue-referenced title, scaffolding removed. '' for a pure reference without a concept ('Bemerkung 7.22' → ''); the concept otherwise ('Satz 7.13 (Christofides)' → 'Christofides', 'Satz 7.6: Kriterium für X' → 'Kriterium für X', 'N P … (Definition 6.19)' → 'N P …'). A title without any catalogue reference comes back unchanged.""" t = (title or "").strip() t = _REF_SUFFIX_RE.sub("", t).strip() m = _REF_PREFIX_RE.match(t) if m: t = t[m.end():].strip() if (w := _WHOLE_PAREN_RE.match(t)): # leading label left a lone '(Christofides)' t = w.group(1).strip() return t def _is_reference(title: str) -> bool: """True for a pure reference/placeholder title WITHOUT concept content: "Satz 7.18", "Lemma 6.2", "Korollar 6.18" (number, no name) and marked spots "Bedingung (**)". False once a concept survives the catalogue strip ("Satz 7.13 (Christofides)", "Satz 6.24: Cook/Levin") and for short technical symbols like "P⊆NP"/"Σ*" (real concepts, no reference scaffolding).""" t = title.strip() if re.search(r'\(\*+\)', t): # marked spot "(**)" / "(*)" return True return bool(_REFERENCE_RE.match(t)) or not _reference_strip(t) def _canonical(candidates: list[dict], idxs: list[int], seen_norm: set[str]) -> dict: """Representative of a cluster = the main concept (fewest property markers — the problem itself, not "… ∈ NP"); tie → most frequent norm title → most readers. Title globally unique (suffix ' (2)') so it works as a key.""" by_norm: dict[str, list[int]] = {} for k in idxs: by_norm.setdefault(_norm_title(candidates[k]["title"]), []).append(k) def weight(nb: str): ms = by_norm[nb] reader = set().union(*[set(candidates[m]["reader"]) for m in ms]) if ms else set() # reference/placeholder titles ("Satz 7.18") last — prefer a meaningful member. is_real = not _is_reference(candidates[ms[0]]["title"]) return (is_real, -_aspect_marker(nb), len(ms), len(reader)) best = max(by_norm, key=weight) k = max(by_norm[best], key=lambda m: len(candidates[m]["description"])) title = candidates[k]["title"] n = 2 while _norm_title(title) in seen_norm: title = f"{candidates[k]['title']} ({n})" n += 1 seen_norm.add(_norm_title(title)) return {"title": title, "description": candidates[k]["description"]} def _pairs_schema(data) -> dict[int, bool] | None: """{"pairs": {"1": "ja", "2": "nein", …}} → {pair_nr: True/False} · otherwise None.""" if not isinstance(data, dict) or not isinstance(data.get("pairs"), dict): return None out: dict[int, bool] = {} for k, v in data["pairs"].items(): try: nr = int(k) except (ValueError, TypeError): continue out[nr] = str(v).strip().casefold() in ("ja", "yes", "true", "1") return out or None def _cliques(n: int, edge_list: list[tuple[int, int]]) -> list[list[int]]: """Complete-link: greedy maximal cliques over the confirmed duplicate edges. A group forms only if ALL its nodes are pairwise connected → no chaining (A=B + B=C forms NO group {A,B,C} as long as A=C is missing). Only cliques ≥2 are returned.""" adj: dict[int, set[int]] = {i: set() for i in range(n)} for a, b in edge_list: adj[a].add(b) adj[b].add(a) used: set[int] = set() groups: list[list[int]] = [] for v in sorted(range(n), key=lambda x: -len(adj[x])): if v in used or not adj[v]: continue clique = {v} for u in sorted(adj[v], key=lambda x: -len(adj[x])): if u not in used and clique <= adj[u] | {u}: # u connected to ALL previous ones clique.add(u) if len(clique) >= 2: groups.append(sorted(clique)) used |= clique return groups # Canonical-name blocking (dedup recall): entity resolution's recall ceiling is set by candidate # generation — a pair that never shares a candidate can never be merged. Normalize a title to a # scaffolding-free, operator-class-normalized, order-independent key so surface variants of ONE entity # collapse ("Offenes Problem P=NP?" ≡ "P vs NP" ≡ "P=NP"). Generic (no course terms): strip a small # stoplist of catalogue/wrapper words, fold the relation operators into class tokens, sort content tokens. _CANON_STOP = re.compile( r'\b(?:offenes?|open|problem|frage|question|algorithmus|algorithm|verfahren|method|methode|' r'satz|theorem|lemma|korollar|definition|def|das|der|die|the|ein|eine|einen|a|an|' r'von|of|für|for|und|and|zum|zur|im)\b', re.I) # German compounds are head-final: "Cliquenproblem" = stem + optional Fuge + head noun "problem". # The standalone wrapper "problem" is already in _CANON_STOP, but a GLUED head hides the stem from ER # blocking ("Cliquenproblem" would never share a key with "Clique"). Strip the glued "problem" head so # the stem surfaces — a generic suffix rule, not a term list. The Fuge is only a consonant n/s # ("Cliquen-", "Entscheidungs-"): matching a vowel "e" would eat the stem's own "e" ("clique" → "cliqu"). # The ≥3-char stem lookbehind keeps a card literally titled "Problem"/"Probleme" (→ _CANON_STOP) intact. _CANON_GLUED_HEAD = re.compile(r'(?<=\w{3})[ns]?problem(?:en|e|s)?\b', re.I) # Catalogue references ("Definition 6.19", "Satz 7.8") are scaffolding INCLUDING their number — # stripped as a phrase, so the digits don't pollute the key. Variant digits ("3-SAT") have no # scaffolding word in front and survive. _CANON_CATALOGUE = re.compile( r'\b(?:definition|def|satz|lemma|korollar|corollary|theorem|proposition' r'|kapitel|chapter|abschnitt|section)\s*\d+(?:\.\d+)*\b', re.I) _PAREN_GROUP = re.compile(r'^\s*(.*?)\s*\(([^()]{2,60})\)\s*$') def _title_variants(title: str) -> set[str]: """Acronym/expansion variants of a "X (Y)" title — normalized outer part and paren content. „Satisfiability Problem (SAT)" → {'satisfiability problem', 'sat'}: matched against another card's norm/key this makes acronym↔expansion pairs dedup CANDIDATES (measured: title cosine 'sat' vs the long form is 0.53, far below the floor). Titles without exactly one trailing paren group → empty set.""" from textkit import _norm_title m = _PAREN_GROUP.match(title or "") if not m: return set() return {v for v in (_norm_title(m.group(1)), _norm_title(m.group(2))) if v} def _canonical_key(title: str) -> str: """Order-independent canonical key of a title (scaffolding stripped, relation operators normalized). Two titles with the same key denote the same entity with ~100% precision (ER blocking). Empty string if nothing survives (never auto-merged).""" s = unicodedata.normalize("NFKC", title) s = re.sub(r'([a-zäöüß])([A-ZÄÖÜ])', r'\1 \2', s) # CamelCase → two tokens s = s.casefold() s = _CANON_CATALOGUE.sub(' ', s) s = re.sub(r'≟|\bversus\b|\bvs\.?\b|=', ' opeq ', s) # equality / "vs" → one token s = re.sub(r'≤|⪯|→|⇒|⟹|\breduces?\s+to\b|\breduziert\b', ' opred ', s) # reduction → one token s = _CANON_GLUED_HEAD.sub(' ', s) # glued "…problem" head → stem s = _CANON_STOP.sub(' ', s) s = re.sub(r'[^\w ]', ' ', s) # drop punctuation/symbols return " ".join(sorted(t for t in s.split() if t)) # Relation-triple individuation (dedup precision): a reduction/relation "A ≤ B" is identified by BOTH # operands AND direction (RDF-triple identity / SKOS narrowMatch — a subset/restriction is NOT the same). # So "SAT ≤ Clique" ≠ "3-SAT ≤ Clique" (source differs) and "A → B" ≠ "B → A" (direction). Two DIFFERENT # relations must never merge, even if a judge or a high title-cosine says so. _REL_STRIP = re.compile(r'^\s*(?:satz|lemma|korollar|corollary|theorem|proposition)\s*[\d.]*\s*:?\s*' r'|^\s*redu[ck]tion(?:en|s)?\s*:?\s*', re.I) # trailing scaffolding ("SetCover ≤ HittingSet Reduktion") is NOT part of the target operand — # without this strip the guard false-alarms and blocks the correct merge with the bare relation _REL_STRIP_TAIL = re.compile(r'[\s\-–]*(?:redu[ck]tion(?:en|s)?|transformation(?:en|s)?)\s*$', re.I) # an attached p/m marker ("≤p", "≤m") is operator notation, not part of the right operand _REL_OPERATOR = re.compile(r'[≤⪯≥⊆⊊→⇒⟹⇔←][pm]?|=>|<=|->', re.I) def _relation_operands(title: str) -> tuple[str, str] | None: """(canonical_source, canonical_target) of a relation/reduction title, else None (not a relation). Operands canonicalized (lowercased, non-alphanumerics stripped) so spacing/hyphenation don't matter.""" t = _REL_STRIP.sub('', title, count=1) t = _REL_STRIP_TAIL.sub('', t, count=1) m = _REL_OPERATOR.search(t) if not m: return None left = re.sub(r'[\W_]', '', t[:m.start()].casefold()) # keep unicode letters/digits (umlauts), drop the rest right = re.sub(r'[\W_]', '', t[m.end():].casefold()) if not left or not right: return None return (left, right) def _relation_conflict(title_a: str, title_b: str) -> bool: """True if BOTH titles are relations/reductions but denote DIFFERENT ones (operands or direction differ) → they must NOT be merged. False if either is not a relation, or they are the same relation.""" a, b = _relation_operands(title_a), _relation_operands(title_b) return a is not None and b is not None and a != b def _direction_conflict(title_a: str, title_b: str) -> bool: """Nur der DETERMINISTISCH sichere Teil des Relation-Guards: dieselben Operanden in GETAUSCHTER Richtung („A ≤ B" vs „B ≤ A") — immer eine andere Reduktion. Ein Operand-Unterschied (z. B. „CLIQUE" vs „k-CLIQUE" — Synonym? oder echt verschieden?) ist eine Bedeutungsfrage und bleibt dem (belegten) Judge überlassen, statt per String-Vergleich blind blockiert zu werden.""" a, b = _relation_operands(title_a), _relation_operands(title_b) return a is not None and b is not None and a != b and set(a) == set(b) def _filter_schema(data) -> dict[int, int] | None: """{"fragments": {"3": 7, "12": 8}} → {block_nr: parent_nr} · None on invalid structure. Empty dict = valid (nothing to degrade). Parent ≠ itself.""" if not isinstance(data, dict) or not isinstance(data.get("fragments"), dict): return None out: dict[int, int] = {} for k, v in data["fragments"].items(): try: nr, parent = int(k), int(v) except (ValueError, TypeError): continue if nr != parent: out[nr] = parent return out # Pure notation/symbols without a standalone concept — kept narrow (FP~0, checked against aak; # "KNF"/"MST"/"NP" do NOT match). These are discarded autonomously (need no parent). _FILTER_NOTATION = re.compile(r'^\s*\|.{1,6}\|\s*$|^Güte\s+\d+\s*$') # Exercise-sheet / cross-reference artefacts — the SECOND gate for a judge `drop` verdict: a hard-drop # (parentless removal) only fires when the judge lists the number in `drop` AND `_is_artifact(title)`. # A judge-drop without a match degrades to "keep" (logged), never deleted (FP~0 discipline like # _FILTER_NOTATION). Shape-matching on a NORMALIZED, title-side string (see _is_artifact): four branches — # (1) lettered/Roman/numeric sub-claims "(Aussage i)"/"Aussage (a)"/"Teil (b)"/"Fall (2)" in ANY # parenthesization; (2) sheet refs "Blatt 10"/"Aufgabe 3"/"Übung"; (3) worked-example/table/figure refs # "Beispiel Tab. 7.1" (a NUMBER is required — guards polysemes "Hash-Tabelle"/"Bijektive Abbildung"); # (4) parenthesized "(Variante)". Theorem words (Satz/Definition/Lemma) are deliberately NOT in the # vocabulary, so "Satz 6.24 Cook/Levin — SAT ist NP-vollständig" is kept. Word boundaries guard prefixes # (Aussagenlogik/Teilmenge/Blattknoten/Fallunterscheidung). _FILTER_ARTIFACT = re.compile(r""" \b(?:aussage|teil|behauptung|fall)\b[\s(]*(?:[ivx]{1,4}|[a-z]|\d{1,2})\)?(?![a-zäöüß]) | \b(?:blatt|aufgabe|serie|hausaufgabe)\s*\d+(?:[.,]\d+)* | \b(?:übungsblatt|übung|uebung)\b | \b(?:beispiel|abbildung|abb|tabelle|tab|bild|grafik|diagramm|figur|skizze)\b\.?\s*\d+(?:[.,]\d+)* | \(\s*(?:variante|variation|spezialfall|sonderfall)\s*\) """, re.VERBOSE) def _is_artifact(title: str) -> bool: """True if the TITLE looks like exercise-sheet / cross-reference scaffolding (P1 hard-drop gate). NFKC + casefold normalization (position/case/umlaut/Unicode-invariant: folds Ⅱ→ii, full-width, NBSP), then match only the title side of the em-dash — a real concept whose *description* merely cites a sheet ("… — vgl. Aufgabe 3") is never dropped. Normalize for matching only; never store the result.""" norm = re.sub(r"\s+", " ", unicodedata.normalize("NFKC", title).casefold()).strip() head = re.split(r"\s[—–-]\s", norm, maxsplit=1)[0] # spaced dash only → "3-SAT"/"np-schwer" unsplit return bool(_FILTER_ARTIFACT.search(head)) # Property/runtime suspicion — marks lines for the judge's verdict (NO auto-drop, FP too high: # "NP-Schwere", reductions with "∈NP" are real blocks). Complements _aspekt_marker. _FILTER_PREDICATE = re.compile( r'ist NP-(vollständig|schwer)|NP-(Vollständigkeit|Schwere) von|ETH (Konsequenz|Lower Bound)' r'|Approximationsschema nach|Laufzeit O\(|∈ ?NP' # fragment families that the aspect substrings miss (all title/desc, advisory ⚠ only): r'|\bSatz\s+\d|\bLemma\s+\d|\bKorollar\s+\d' # bare theorem/proof references (with number) r'|^\s*(?:Remark|Bemerkung|Anmerkung|Note|Notiz|Beobachtung|Observation)\b' # EN+DE remark labels r'|\bSatz\s*:|\bSatz\s+[A-Z]\b' # "Satz:" (colon, no number) / "Satz D*" letter label r'|\bGegenbeispiel\b|\bWorst[- ]?Case\b|Schärfe\s+der\b' # proof-example / sharpness facets r'|2\s*\^\s*[{(]?\s*[Ωωoο]\s*\(' # ETH exponential bound 2^Ω(…)/2^o(…) r'|\d\s*[−–-]\s*1\s*/\s*m' # approximation-güte ratio "2 − 1/m" r'|Variablenungleichung|[α-ωΑ-Ω][a-z]?-?Variablen', re.I) # proof variables (αu-Variablen …) def _filter_suspect(b: dict) -> bool: """Heuristic flag: could be a property/detail of another block.""" return _aspect_marker(b["title"]) > 0 or bool(_FILTER_PREDICATE.search(f"{b['title']} {b['description'] or ''}")) def _root(nr: int, fragments: dict[int, int]) -> tuple[int, bool]: """Follow the fragment→parent chain to the first ancestor that is NOT itself a fragment. Returns (root, cyclic). This walk IS the fixpoint (no separate iteration): a mid-tier fragment whose own parent is also a fragment resolves to the top-level real block, so all chain levels collapse in one pass (unlike the old single-level parent_set shield). Cycle guard: on revisiting a node returns (node, True) → caller keeps both (no annihilation).""" seen: set[int] = set() cur = nr while cur in fragments: if cur in seen: return cur, True seen.add(cur) cur = fragments[cur] return cur, False def _containment_parent(frag_norm: str, others: list[tuple[int, str]]) -> int | None: """Deterministic parent-by-name-containment for a ⚠-flagged survivor: a fragment whose title NAMES another block ("Lower Bound … für VERTEX COVER" → Vertex Cover; "List Scheduling Güte …" → List Scheduling). `others` = (nr, title_norm) of the OTHER blocks. A parent is a block whose normalized title occurs as a WHOLE-WORD span inside `frag_norm`, is SIGNIFICANT (≥2 tokens or ≥6 chars — never "p"/"np"/"sat") and strictly shorter than the fragment. Returns the parent nr on EXACTLY ONE match, else None (0 or ≥2 → leave to the judge). Whole-word + significance + exactly-one → near-FP-0.""" hits = [] for nr, ptitle in others: if not ptitle or ptitle == frag_norm or len(ptitle) >= len(frag_norm): continue if len(ptitle) < 6 and ptitle.count(" ") < 1: # reject short single-token names (P/NP/SAT) continue if re.search(r'(? bool: """True for the narrow, parent-less noise classes safe to hard-drop (subject to KEEP-guards).""" norm = re.sub(r"\s+", " ", unicodedata.normalize("NFKC", title).casefold()).strip() head = re.split(r"\s[—–-]\s", norm, maxsplit=1)[0] return bool(_PARENTLESS_NOISE.search(head)) and not _PARENTLESS_KEEP.search(norm) # F1 — statement-gate keep-guard (Knowledge-Component / OMDoc theory): a self-contained ASSERTION is its # own learning unit, not a whole-part fragment. Protect two general classes from demotion: # (a) a named reduction between two PROBLEMS ("3-SAT ≤ Clique", "Clique → Vertex Cover"), and # (b) a LABELED or ATTRIBUTED theorem carrying its own biconditional/implication ("Satz 6.37: … ⇔ …"). # NOT protected: a bare label ("Satz 7.18", "Remark 7.28"), unary status ("X ist NP-vollständig", "X ∈ NP"), # a proof-size step ("Strikte Reduktion |A| = O(m)"), or a güte/bound facet — those carry neither a # two-sided reduction operator NOR a ⇔/⇒ assertion. General: attribution is structural, no author whitelist. _STMT_LABEL = re.compile(r'^\s*(?:satz|lemma|korollar|theorem|proposition|prop|folgerung)\b', re.I) _STMT_ATTRIB = re.compile(r'\b(?:satz|lemma|theorem|korollar)\s+von\s+[A-ZÄÖÜ]') # "Satz von Cook/Levin" _STMT_ASSERT = re.compile(r'⇔|⇒|⟺|⟹|\bgdw\.?\b|\bgenau dann\b', re.I) _REDUCTION_ONLY_OP = re.compile(r'[≤⪯]|→|⇒|⟹|->') # genuine reduction operators (NOT plain "=") def _is_reduction_statement(title: str) -> bool: """True if the title is a reduction between two NAMED problems (both sides carry ≥3 letters and neither is a pure bound like "O(m)"). Rejects unary "X ∈ NP", proof-size "|A| = O(m)", and a construction/assignment suffix after the target ("3-SAT ≤ K-COLOR: G=(V,E) Konstruktion" is a construction fragment, not a standalone statement — a bare target carries no ':' or '=').""" t = _REL_STRIP.sub('', title, count=1) m = _REDUCTION_ONLY_OP.search(t) if not m: return False right = t[m.end():] if ':' in right or '=' in right: # construction/assignment detail after the target → fragment return False def _named(s: str) -> bool: return len(re.findall(r'[a-zäöüß]', s, re.I)) >= 3 and not re.match(r'\s*[Oo]\s*\(', s) return _named(t[:m.start()]) and _named(right) def _is_named_statement(title: str, desc: str = "") -> bool: """Statement-gate keep-guard: a named reduction (a) or a labeled/attributed theorem WITH its own ⇔/⇒ assertion (b). Bare labels / unary status / proof-size steps return False (stay demotable).""" if _is_reduction_statement(title): return True if (_STMT_LABEL.match(title) or _STMT_ATTRIB.search(title)) and _STMT_ASSERT.search(f"{title} {desc or ''}"): return True return False def _umbrella_schema(data, ids: set[int]): """{"umbrellas":[{"title":str,"description":str,"members":[int,…]}, …]} → [(title, description, [member ids])]. [] = valid (no umbrella); None ONLY on broken JSON (so resume treats a valid-but-empty file as done, like _filter_schema's {} vs None). Each id used at most once across all umbrellas (first wins); members filtered to `ids`; an umbrella needs ≥2 surviving members; `description` required + non-empty — it MUST enumerate the children, else the source-scoped subblock step can't re-derive them (no web on uni).""" if not isinstance(data, dict) or not isinstance(data.get("umbrellas"), list): return None out, used = [], set() for u in data["umbrellas"]: if not isinstance(u, dict): continue title = str(u.get("title", "")).strip() desc = str(u.get("description", "")).strip() raw_members = u.get("members") if not title or not desc or not isinstance(raw_members, list): continue members = [] for x in raw_members: try: m = int(x) except (ValueError, TypeError): continue if m in ids and m not in used: used.add(m) members.append(m) if len(members) >= 2: out.append((title, desc, members)) return out def _completion_schema(data, n_umbrellas: int, ids: set[int]): """{"additions":[{"umbrella":int,"members":[int,…]}, …]} → [(umbrella_idx, [member ids])]. [] = valid (nothing to absorb); None only on broken JSON. umbrella idx in range; members drawn from `ids` (the still-standalone leftovers), de-duped within an addition.""" if not isinstance(data, dict) or not isinstance(data.get("additions"), list): return None out = [] for a in data["additions"]: if not isinstance(a, dict): continue try: k = int(a.get("umbrella")) except (ValueError, TypeError): continue if not (0 <= k < n_umbrellas): continue mem = [] for x in (a.get("members") or []): try: m = int(x) except (ValueError, TypeError): continue if m in ids and m not in mem: mem.append(m) if mem: out.append((k, mem)) return out # Deterministic backstop to the grouping judge's TEST 1 (type gate): an umbrella may bundle ONLY # constituent sub-definitions of ONE definition. If a member title carries a standalone-unit signal # (a named algorithm / problem / reduction / theorem), the umbrella is dissolved — those stay their own # blocks. Kept narrow so real definition-parts (Konfiguration, Übergangsfunktion δ, Literale, Makespan, # m Maschinen) never match; checked against the aak over-merge (member „Greedy-Algorithmus GA" hits). # Suffix-anchored head nouns (German compounds are head-final: „Approximations+algorithmus" has NO word # boundary before „algorithmus", so \bAlgorithmus\b misses it → the MAX-SAT over-merge). \w* absorbs the # modifier; the head noun stays the discriminator. FP-safe: no real TM/KNF definition-part ends in these # heads (Berechnung is deliberately NOT a head → „Akzeptierende Berechnung" stays a valid member). _GROUP_STANDALONE = re.compile( r'\w*algorithm(?:us|en)\b|\w*problem(?:e|s|en)?\b|\w*reduktion(?:en)?\b|\bscheduling\b|[≤⪯]' r'|^\s*(?:Satz|Lemma|Korollar|Theorem|Bemerkung|Beobachtung)\s*\d' # atomicity: a named COMPLEXITY CLASS / a "…-Vollständigkeit(completeness)" / a "…Transformation" is a # self-contained concept (learning-object / atomic-KC), never a sub-definition — a bundle of ≥1 such # member is siblings, not one model → dissolve (catches the P/NP/NP-Vollständigkeit over-merge that NO # cosine floor separates). Head-final compounds (\w*klasse absorbs "Komplexitäts+klasse"); FP-safe — # no real TM/KNF/TSP/Scheduling definition-part carries these heads. r'|\w*vollständigkeit\b|\w*completeness\b|\w*transformation(?:en)?\b|\w*klasse[nr]?\b', re.I) # --- Outline (blocks artifact: chapter structure, only read by the guide) --- def _outline_review_schema(data, valid: set[int], n_chapters: int, n_blocks: int): """{"moves": {"": }} → {nr: idx} (may be {}) · None if broken/invalid. A mass rewrite (more than a third of all blocks) is rejected — the reviewer's job is spotting misplacements, not re-designing the outline.""" if not isinstance(data, dict) or not isinstance(data.get("moves"), dict): return None out: dict[int, int] = {} for k, v in data["moves"].items(): try: nr, ch = int(k), int(v) except (ValueError, TypeError): return None if nr not in valid or not (1 <= ch <= n_chapters): return None out[nr] = ch if len(out) * 3 > n_blocks: return None return out def _outline_schema(data, valid: set[int]): """{"chapters":[{title,numbers}]} → cleaned (valid numbers, each exactly once) · None at <80 % coverage (agent/judge omitted too much).""" if not isinstance(data, dict) or not isinstance(data.get("chapters"), list): return None out, seen = [], set() for ch in data["chapters"]: if not isinstance(ch, dict): continue title = str(ch.get("title", "")).strip() or "Chapter" nums = [] for n in (ch.get("numbers") or []): try: n = int(n) except (ValueError, TypeError): continue if n in valid and n not in seen: seen.add(n) nums.append(n) if nums: out.append({"title": title, "numbers": nums}) if not out or len(seen) < 0.8 * len(valid): return None return {"chapters": out} def _prereq_schema(data, valid: set[int]) -> dict[int, list[int]]: """{"prereqs": {"3": [1, 7]}} → {num: [prereq nums]} · only numbers from `valid`, no self-edge. Invalid/empty → {} (best-effort: then original order).""" if not isinstance(data, dict) or not isinstance(data.get("prereqs"), dict): return {} out: dict[int, list[int]] = {} for k, v in data["prereqs"].items(): try: num = int(k) except (ValueError, TypeError): continue if num not in valid or not isinstance(v, list): continue pres = [] for p in v: try: p = int(p) except (ValueError, TypeError): continue if p in valid and p != num and p not in pres: pres.append(p) if pres: out[num] = pres return out def _topo_order(nums: list[int], edges: dict[int, list[int]]) -> list[int]: """Kahn topo sort: prerequisites first. `edges[num]` = numbers that must come BEFORE num. Stable tie-break (original order of `nums`); cycles are broken (never deadlock).""" pos = {n: i for i, n in enumerate(nums)} # remaining in-degree over valid nodes only; self/foreign edges ignored. pre = {n: [p for p in edges.get(n, []) if p in pos and p != n] for n in nums} done: list[int] = [] finished: set[int] = set() rest = list(nums) while rest: ready_nodes = [n for n in rest if all(p in finished for p in pre[n])] if not ready_nodes: # cycle → force the earliest remaining node in original order ready_nodes = [min(rest, key=lambda n: pos[n])] nxt = min(ready_nodes, key=lambda n: pos[n]) # stable: smallest original position first done.append(nxt) finished.add(nxt) rest.remove(nxt) return done async def _learning_order(ctx: GenContext, set_p, files: dict, entries: dict, valid: set[int], instructions: str) -> dict: """Put entries (num→title) into learning order: the LLM extracts prereq edges from the extracted `prerequisites`, code solves via topo sort. Best-effort → otherwise entries unchanged.""" if len(entries) < 3: return entries topic = ctx.topic facts_map = _json_file(files["facts"]) facts_map = facts_map if isinstance(facts_map, dict) else {} def _hint(title): fm = facts_map.get(title) or {} vs = [v for fk in fm.values() if isinstance(fk, dict) and (v := str(fk.get("prerequisites", "")).strip())] return " · ".join(dict.fromkeys(vs)) pp = files["arbeit"] / "outline-prereqs.json" def _payload(result, p=pp): d = _sink_or_file(result, p, lambda x: x if isinstance(x, dict) and "prereqs" in x else None) return d existing = _json_file(pp) if not (isinstance(existing, dict) and "prereqs" in existing): lines = [f"{n}. {t}" + (f"\n braucht vorher: {h}" if (h := _hint(t)) else "") for n, t in entries.items()] set_p("Outline — learning order…", step=_step_idx(topic, "Outline")) await run_single_slot( ctx, "Outline-Prerequisites", key=f"blocks-{topic}-outline-prereqs", prompt=_prompt("Outline-Prerequisites", topic=topic, blocks="\n".join(lines), out_path=pp, extra=_extra(instructions)), role="guide", capabilities="files", payload=_payload, timeout=_timeout("plan", len(entries))) edges = _prereq_schema(_json_file(pp), valid) if not edges: return entries # no/invalid edges → original order (no regression) ordered = _topo_order(list(entries), edges) return {n: entries[n] for n in ordered} async def _outline_block(ctx: GenContext, set_p, files: dict, entries: dict, instructions: str) -> dict: """Format-agnostic outline over ALL blocks — 3 proposals → judge merges. Never aborts: 0 valid → one chapter with everything; missing blocks land in "Other". → {"chapters":[{title,numbers}]} (also in files["outline"]).""" topic, is_cancelled = ctx.topic, ctx.is_cancelled valid = set(entries) step = _step_idx(topic, "Outline") # Establish learning order (LLM-modulo): the LLM extracts prereq edges from the extracted # `prerequisites`, code solves via topo sort. Best-effort → otherwise original order. entries = await _learning_order(ctx, set_p, files, entries, valid, instructions) liste = "\n".join(f"{n}. {t}" for n, t in entries.items()) set_p("Outline — proposals…", step=step) async def _proposal(i, path): if _outline_schema(_json_file(path), valid): return True await run_single_slot( ctx, f"Outline {i}", key=f"blocks-{topic}-outline-{i}", prompt=_prompt("Guide-Outline", topic=topic, blocks=liste, out_path=path, extra=_extra(instructions)), role="guide", capabilities="files", payload=lambda result, p=path: _sink_or_file(result, p, lambda d: _outline_schema(d, valid)), timeout=_timeout("plan", len(entries))) return _outline_schema(_json_file(path), valid) is not None slots = files["outline_slots"] await _gather_progress([_proposal(i, p) for i, p in enumerate(slots, 1)], len(slots), _report_p(set_p, topic, "Outline")) if is_cancelled(): return {} proposals = [v for p in slots if (v := _outline_schema(_json_file(p), valid))] if not proposals: plan = {"chapters": [{"title": "Contents", "numbers": list(entries)}]} elif len(proposals) == 1: plan = proposals[0] else: set_p("Outline merging…", step=step) block_texts = "\n\n".join( f"### Vorschlag {i}\n" + "\n".join( f"KAPITEL: {ch['title']}\n Nummern: {', '.join(str(n) for n in ch['numbers'])}" for ch in v["chapters"]) for i, v in enumerate(proposals, 1)) await run_single_slot( ctx, "Outline-Judge", key=f"blocks-{topic}-outline-judge", prompt=_prompt("Guide-Outline-Judge", topic=topic, format_name="den Guide", purpose="alle Blocks in einem roten Faden", n=len(proposals), blocks=liste, outlines=block_texts, out_path=files["outline"], extra=_extra(instructions)), role="judge", capabilities="files", payload=lambda result: _sink_or_file(result, files["outline"], lambda d: _outline_schema(d, valid)), timeout=_timeout("plan_judge", len(entries))) plan = _outline_schema(_json_file(files["outline"]), valid) or proposals[0] # Placement review (best-effort): ONE judge checks every block→chapter assignment and # reports ONLY misplacements as moves. Invalid/mass output → plan unchanged. if proposals and len(plan["chapters"]) >= 2 and not is_cancelled(): rp = files["arbeit"] / "outline-review.json" moves = _outline_review_schema(_json_file(rp), valid, len(plan["chapters"]), len(entries)) if moves is None: chapter_text = "\n\n".join( f"KAPITEL {k}: {ch['title']}\n" + "\n".join(f" {n}. {_title(entries[n])}" for n in ch["numbers"]) for k, ch in enumerate(plan["chapters"], 1)) set_p("Outline review…", step=step) await run_single_slot( ctx, "Outline-Review", key=f"blocks-{topic}-outline-review", prompt=_prompt("Guide-Outline-Review", topic=topic, chapters=chapter_text, out_path=rp, extra=_extra(instructions)), role="judge", capabilities="files", payload=lambda result: _sink_or_file(result, rp, lambda d: _outline_review_schema( d, valid, len(plan["chapters"]), len(entries))), timeout=_timeout("plan_judge", len(entries))) moves = _outline_review_schema(_json_file(rp), valid, len(plan["chapters"]), len(entries)) for nr, target in (moves or {}).items(): for ch in plan["chapters"]: if nr in ch["numbers"]: ch["numbers"].remove(nr) plan["chapters"][target - 1]["numbers"].append(nr) if moves: plan["chapters"] = [ch for ch in plan["chapters"] if ch["numbers"]] _log(topic, f"Outline-Review: {len(moves)} Block/Blöcke umsortiert") # Completeness: every block appears — missing in "Other" (against omitting agents/judge). included = {n for ch in plan["chapters"] for n in ch["numbers"]} missing = [n for n in entries if n not in included] if missing: plan["chapters"].append({"title": "Other", "numbers": missing}) atomic_write_json(files["outline"], plan, indent=1) return plan # --- Learning artefacts (flashcards/examples from the facts) --- async def _mirror_sidecar_db(topic: str, sidecar: dict) -> None: """Mirror the sidecar {block title: [{title, level, relevance}]} into the DB table subblocks.""" for btitle, subs in sidecar.items(): bnorm = _norm_title(btitle) if not bnorm or not isinstance(subs, list): continue for s in subs: if not isinstance(s, dict): continue st = str(s.get("title", "")).strip() sn = _norm_title(st) if not sn: continue facts = json.dumps(s["facts"], ensure_ascii=False) if isinstance(s.get("facts"), dict) else None await db.put_subblock(topic, bnorm, sn, btitle, st, level=s.get("level"), relevance=s.get("relevance"), facts=facts, status="consensus") async def _guide_ebene(topic: str, instructions: str, provider: str, is_cancelled, set_p) -> None: """Guide-Ebene der Auto-Sequenz: generieren → QA → „Befunde beheben"-Loop bis 100 %/Stillstand/10×. Fail-open: Fehler überspringen den Guide, blockieren nichts.""" import uuid from datetime import datetime, timezone try: import guide, guide_qa, guide_board from auto_loop import auto_repair_loop fmt = "Guide" now = datetime.now(timezone.utc).isoformat() guide_id = str(uuid.uuid4()) await db.create_guide({"id": guide_id, "topic": topic, "format": fmt, "instructions": instructions, "status": "queued", "progress": None, "created_at": now, "updated_at": now}) set_p("Guide wird generiert…") await guide.generate_guide(guide_id, topic, fmt, instructions, provider) if is_cancelled(): return async def _messen(): rep = await guide_qa.guide_qa_report(topic, llm=True) return float(rep["note_guide"]) if rep else 10.0 erst = await _messen() async def _reparieren(): set_p("Befunde beheben (Guide)…") await guide_board.repair_karten(topic, fmt) # befundtragende Karten → pruefer await guide.generate_guide(topic=topic, format_name=fmt, guide_id=guide_id, instructions=instructions, provider=provider) # resumt offene Karten return await _messen() res = await auto_repair_loop("Guide", erst, _reparieren) if res["grund"] != "fertig": msg = (f"Guide bleibt bei {round(res['note'] * 10)} % ({res['grund']}, " f"{res['runden']} Runden) — «Befunde beheben» hat keinen Pfad (Systemfehler)") _blocks_errors[topic] = msg set_p(msg) else: set_p(f"Guide: 100 % nach {res['runden']} Runde(n)") except Exception as e: log.exception("[%s] Guide-Ebene fehlgeschlagen", topic) _blocks_errors[topic] = f"Guide-Ebene: {str(e)[:500]}" async def generate_blocks(topic: str, instructions: str = "", provider: str = DEFAULT_PROVIDER, research: bool = True, qa_force: bool = False, auto_inventory: bool = True, auto_artefacts: bool = True, auto_guide: bool = True) -> None: """Kanban entry point: source prep, then both boards (inventory + artefacts) until quiescence. research=False = Continue (drain the existing queue, no new search). A run on a finished topic ADDS research (live extension) — full rebuild = DELETE /blocks. auto_* per level: after each level's QA, loop „Befunde beheben" until 100 % / stall / 10×; if off, stop after that level's QA (control point).""" if topic in _blocks_progress: return _blocks_progress[topic] = "Warten…" _blocks_errors.pop(topic, None) files = _blocks_files(topic) q = load_source(topic) folder = source_folder(topic) # projekt/uni/link → folder, thema → None instructions = q.get("spec") or instructions # prefer the persisted specification (also on resume) def set_p(msg: str, step: int | None = None) -> None: _blocks_progress[topic] = msg if step is not None: _blocks_step[topic] = step def is_cancelled() -> bool: return topic in _blocks_cancelled ctx = GenContext(topic=topic, provider=provider, is_cancelled=is_cancelled) try: async with _semaphore: files["arbeit"].mkdir(parents=True, exist_ok=True) # Step "Source prep": crawl (link) + PDFs + content/noise triage. if not await _prepare_source(ctx, set_p, files, q, folder, instructions): if is_cancelled(): _blocks_errors[topic] = "Cancelled — progress is preserved" return import board_inventory # lazy: the boards import blocks status: dict = {} ok = await board_inventory.run_boards(ctx, set_p, files, q, folder, instructions, research=research, qa_force=qa_force, auto_inventory=auto_inventory, auto_artefacts=auto_artefacts, status_out=status) if not ok and is_cancelled(): _blocks_errors[topic] = "Cancelled — progress is preserved" elif ok and auto_guide and status.get("weiter") and not is_cancelled(): # Auto-Sequenz: Inventar + Artefakte liefen sauber durch → Guide-Ebene. await _guide_ebene(topic, instructions, provider, is_cancelled, set_p) except Exception as e: log.exception("[%s] Blocks generation failed", topic) _blocks_errors[topic] = str(e)[:2000] finally: # No file cleanup: intermediate files stay for resume / traceability. _blocks_progress.pop(topic, None) _blocks_step.pop(topic, None) _blocks_cancelled.discard(topic) clear_scope(f"blocks-{topic}-") # clear the scope → restart isn't blocked