This commit is contained in:
team3
2026-07-06 14:44:20 +02:00
parent f4c5116abb
commit cc27e53b9e
20 changed files with 504 additions and 267 deletions

View File

@@ -24,11 +24,11 @@ from pathlib import Path
import database as db import database as db
import embedding import embedding
from agents import kill_process, cancel_scope, clear_scope from agents import kill_process, cancel_scope, clear_scope
from config import CONSENSUS_GRACE, CONSENSUS_MAX_ROUNDS, DEFAULT_PROVIDER, CRAWL_KEEP_PATTERNS, CRAWL_NOISE_PATTERNS, CRAWL_MIN_CHARS, QUELLE_RELEVANZ_CHUNK, QUELLE_RELEVANZ_SNIPPET, EMBEDDING_AKTIV, EMBEDDING_SUB_DUP, SUB_VARIANT_COS, SUBBLOCK_MAX, EVIDENCE_BUDGET_CHARS, EVIDENCE_CTX_LINES from config import CONSENSUS_GRACE, CONSENSUS_MAX_ROUNDS, DEFAULT_PROVIDER, EMBEDDING_AKTIV, EMBEDDING_SUB_DUP, SUB_VARIANT_COS, SUBBLOCK_MAX, EVIDENCE_BUDGET_CHARS, EVIDENCE_CTX_LINES
from fsutil import atomic_write_json from fsutil import atomic_write_json
from jsonio import parse_json_text, read_json_file as _json_file from jsonio import parse_json_text, read_json_file as _json_file
from paths import arbeit_dir, blocks_path, question_pattern_path, project_dir, subblocks_path, source_path, source_crawl_dir, safe_folder from paths import arbeit_dir, blocks_path, question_pattern_path, project_dir, subblocks_path, source_path, source_crawl_dir, safe_folder
from crawl import crawl from crawl import load_pages
from pipeline import ( from pipeline import (
GenContext, _extra, _gather_progress, _yesno_schema, _log, _prompt, _race, GenContext, _extra, _gather_progress, _yesno_schema, _log, _prompt, _race,
_semaphore, _timeout, run_single_slot, _semaphore, _timeout, run_single_slot,
@@ -853,117 +853,10 @@ def _crawl_index(folder) -> dict[str, str]:
return idx return idx
def _triage_rules(folder, pages: list[str]) -> tuple[list[str], list[str]]:
"""Deterministic content/noise filter (config.CRAWL_*). Substring match (lowercase) against
URL + filename. Order: keep > noise > min_chars > keep. → (content, noise)."""
folder = Path(folder)
content, noise = [], []
for fn in pages:
lines = _read(folder / fn).splitlines()
url = lines[0][len("QUELLE:"):].strip() if lines and lines[0].startswith("QUELLE:") else ""
body = "\n".join(lines[1:]).strip()
hay = f"{url}\n{fn}".lower()
if any(p in hay for p in CRAWL_KEEP_PATTERNS):
content.append(fn)
elif any(p in hay for p in CRAWL_NOISE_PATTERNS):
noise.append(fn)
elif len(body) < CRAWL_MIN_CHARS:
noise.append(fn)
else:
content.append(fn) # default: keep — everything with content stays
return content, noise
def _page_snippet(folder, fn: str) -> tuple[str, str]:
"""(url, snippet) of a crawl page for the relevance gate. url from the QUELLE: line;
snippet = body excerpt (navigation boilerplate is up front — the prompt ignores it).
The URL is the primary signal (meaningful slug), the snippet only supports it."""
lines = _read(Path(folder) / fn).splitlines()
url = lines[0][len("QUELLE:"):].strip() if lines and lines[0].startswith("QUELLE:") else ""
body = "\n".join(lines[1:]).strip()
snippet = " ".join(body.split())[:QUELLE_RELEVANZ_SNIPPET]
return (url or fn), snippet
async def _relevance_triage(ctx: GenContext, set_p, files: dict, folder, content: list[str], spec: str, instructions: str) -> tuple[list[str], list[str]]:
"""LLM topic gate after the rule filter: each content page ja/nein against the spec.
Off-topic (different field) → out. Pattern like `_relevance_block`: small packages, 3 raters
(`fast`), 2-of-3 consensus. CONSERVATIVE: drop only on a clear "nein" majority; dispute/gap/
race error → keep. SAFETY: if the gate would drop ≥80 % (or all), everything stays
(a spec mismatch/bug must not empty the source). → (kept, out) as filenames."""
topic, provider, is_cancelled = ctx.topic, ctx.provider, ctx.is_cancelled
work_dir = files["arbeit"]
pages = sorted(content)
if not pages:
return content, []
items = [_page_snippet(folder, fn) for fn in pages] # index aligns with `pages`
chunks = _chunk_nums(list(range(len(pages))), _n_chunks(len(pages), QUELLE_RELEVANZ_CHUNK))
n = len(chunks)
def rater_paths(c):
return [work_dir / f"source-relevance-c{c}-{i}.json" for i in (1, 2, 3)]
def lset(idxs):
return set(range(1, len(idxs) + 1))
async def _rate(c, idxs):
local_set = lset(idxs)
paths = rater_paths(c)
existing = sum(1 for p in paths if _yesno_schema(_json_file(p), local_set))
if existing >= 2:
return True
enum_lines = []
for k, j in enumerate(idxs, 1):
url, snip = items[j]
enum_lines.append(f"{k}. {url}")
if snip:
enum_lines.append(f" {snip}")
enum = "\n".join(enum_lines)
pending = [(i, p) for i, p in enumerate(paths, 1) if not _yesno_schema(_json_file(p), local_set)]
slots = [{
"key": f"blocks-{topic}-source-relevance-c{c}-{i}",
"prompt": _prompt("Source-Relevance", topic=topic, spec=spec, pages=enum, out_path=p, extra=_extra(instructions)),
"role": "fast", "capabilities": "files",
"payload": (lambda result, p=p, ids=local_set: _yesno_schema(_json_file(p), ids)),
} for i, p in pending]
new = await _race(topic, f"Relevance triage package {c}", slots, 2 - existing, _timeout("relevance", len(idxs)), provider, cancelled=is_cancelled, grace=CONSENSUS_GRACE)
return not is_cancelled() and new is not None
_qidx = _step_idx(topic, "Source prep") # gate runs in the source step (no own step)
set_p(f"Check relevance against spec ({n} packages)…", step=_qidx)
async def _report_triage(d, t):
set_p(f"Check relevance against spec {d}/{t}", step=_qidx)
await _gather_progress([_rate(c, idxs) for c, idxs in enumerate(chunks, 1)], n, _report_triage)
if is_cancelled():
return content, [] # cancel → drop nothing (caller aborts)
# Vote per page: only a clear "nein" majority (≥2 and more than "ja") throws it out.
dropped: list[str] = []
for c, idxs in enumerate(chunks, 1):
local_set = lset(idxs)
rater = [d for p in rater_paths(c) if (d := _yesno_schema(_json_file(p), local_set))]
for k in range(1, len(idxs) + 1):
vote_list = [d[k] for d in rater if k in d]
nein, ja = vote_list.count("nein"), vote_list.count("ja")
if nein >= 2 and nein > ja:
dropped.append(pages[idxs[k - 1]])
if dropped and len(dropped) >= max(1, int(len(pages) * 0.8)):
_log(topic, f"Relevance triage: would drop {len(dropped)}/{len(pages)} — discarded (spec mismatch?), keeping all")
return content, []
dropped_set = set(dropped)
keepers = [fn for fn in pages if fn not in dropped_set]
return keepers, dropped
async def _prepare_source(ctx: GenContext, set_p, files: dict, q: dict, folder, instructions: str) -> bool: async def _prepare_source(ctx: GenContext, set_p, files: dict, q: dict, folder, instructions: str) -> bool:
"""Step "Source prep": crawl (link) + PDF convert + content/noise triage. """Step "Source prep": load link pages + PDF convert. → True (ok) / False (cancel/error).
Persists the triage in the coverage table (content). → True (ok) / False (cancel/error). link: load each user-supplied URL (one page, no following); ALL pages count as content
thema: nothing. projekt/uni: only PDFs (curated folder, no triage).""" (curated — no triage). thema: nothing. projekt/uni: only PDFs (curated folder)."""
topic, is_cancelled = ctx.topic, ctx.is_cancelled topic, is_cancelled = ctx.topic, ctx.is_cancelled
if not folder: if not folder:
return True # thema → no source to prepare return True # thema → no source to prepare
@@ -973,28 +866,21 @@ async def _prepare_source(ctx: GenContext, set_p, files: dict, q: dict, folder,
if await db.get_step_status(topic, "Source prep") == "done": if await db.get_step_status(topic, "Source prep") == "done":
return True return True
if not _crawl_done(topic): if not _crawl_done(topic):
set_p("Loading source (crawl)…", step=_step_idx(topic, "Source prep")) urls = [ln.strip() for ln in q["location"].splitlines() if ln.strip()]
n = await asyncio.to_thread(crawl, q["location"], folder, cancelled=is_cancelled) set_p("Loading pages…", step=_step_idx(topic, "Source prep"))
n = await asyncio.to_thread(load_pages, urls, folder, cancelled=is_cancelled)
if is_cancelled(): if is_cancelled():
return False return False
if not n: if not n:
_blocks_errors[topic] = "Crawl yielded no content — check link/domain" _blocks_errors[topic] = "Links yielded no content — check the URLs"
return False return False
await asyncio.to_thread(_convert_pdfs, folder) await asyncio.to_thread(_convert_pdfs, folder)
pages = sorted(set(_crawl_index(folder).values())) pages = sorted(set(_crawl_index(folder).values()))
if pages: if pages:
set_p("Triaging pages…", step=_step_idx(topic, "Source prep")) # Curated links: the user picked them — keep every page, no rule/LLM triage.
await db.delete_coverage(topic) await db.delete_coverage(topic)
content, noise = _triage_rules(folder, pages) # deterministic rule filter await db.mark_content(topic, pages, [])
if q.get("spec") and content: # topic gate: separates the field (rules can't) _log(topic, f"Links: {len(pages)} pages as content (curated, no triage)")
content, dropped = await _relevance_triage(ctx, set_p, files, folder, content, q["spec"], instructions)
if is_cancelled():
return False
if dropped:
noise = sorted(set(noise) | set(dropped))
_log(topic, f"LLM relevance: {len(dropped)} pages off-topic → noise")
await db.mark_content(topic, sorted(content), sorted(noise))
_log(topic, f"Triage: {len(content)} content / {len(noise)} noise of {len(pages)} (rules + LLM gate)")
await db.set_step_status(topic, "Source prep", "done") await db.set_step_status(topic, "Source prep", "done")
return True return True
@@ -1021,17 +907,41 @@ def _aspect_marker(title: str) -> int:
_REFERENCE_RE = re.compile(r'^(Satz|Lemma|Korollar|Bemerkung|Definition)\s*[\d.]+\s*(\([a-z]\)|[a-z])?\s*$', re.I) _REFERENCE_RE = re.compile(r'^(Satz|Lemma|Korollar|Bemerkung|Definition)\s*[\d.]+\s*(\([a-z]\)|[a-z])?\s*$', re.I)
# Catalogue scaffolding around a concept name: a leading label+number ("Satz 7.13", "Satz 7.6:")
# or a trailing "(Definition 6.19)". _reference_strip peels it off and returns the concept remainder;
# _is_reference is then just "nothing survives the strip". Generic — no author/domain whitelist. The
# remainder feeds the rename step (a card must be titled by its concept, not its catalogue number).
_REF_WORDS = (r'(?:Satz|Lemma|Korollar|Corollary|Bemerkung|Beobachtung|Definition|Def'
r'|Theorem|Proposition|Folgerung|Kapitel|Abschnitt)')
_REF_PREFIX_RE = re.compile(rf'^\s*{_REF_WORDS}\s*\d+(?:\.\d+)*[a-z]?\s*[:.\-–—]?\s*', re.I)
_REF_SUFFIX_RE = re.compile(rf'\s*[\(\[]\s*{_REF_WORDS}\s*\d+(?:\.\d+)*[a-z]?\s*[\)\]]\s*$', re.I)
_WHOLE_PAREN_RE = re.compile(r'^\s*[\(\[]\s*(.{2,}?)\s*[\)\]]\s*$')
def _reference_strip(title: str) -> str:
"""Concept remainder of a catalogue-referenced title, scaffolding removed. '' for a pure reference
without a concept ('Bemerkung 7.22'''); the concept otherwise ('Satz 7.13 (Christofides)'
'Christofides', 'Satz 7.6: Kriterium für X''Kriterium für X', 'N P … (Definition 6.19)'
'N P …'). A title without any catalogue reference comes back unchanged."""
t = (title or "").strip()
t = _REF_SUFFIX_RE.sub("", t).strip()
m = _REF_PREFIX_RE.match(t)
if m:
t = t[m.end():].strip()
if (w := _WHOLE_PAREN_RE.match(t)): # leading label left a lone '(Christofides)'
t = w.group(1).strip()
return t
def _is_reference(title: str) -> bool: def _is_reference(title: str) -> bool:
"""True for pure reference/placeholder titles WITHOUT meaningful content: "Satz 7.18", "Lemma 6.2", """True for a pure reference/placeholder title WITHOUT concept content: "Satz 7.18", "Lemma 6.2",
"Korollar 6.18" (number without a name) as well as marked spots "Bedingung (**)". NOT "Satz 6.24: "Korollar 6.18" (number, no name) and marked spots "Bedingung (**)". False once a concept survives
Cook/Levin" (has a name) and NOT short technical symbols like "P⊆NP"/"Σ*" (real concepts).""" the catalogue strip ("Satz 7.13 (Christofides)", "Satz 6.24: Cook/Levin") and for short technical
symbols like "P⊆NP"/"Σ*" (real concepts, no reference scaffolding)."""
t = title.strip() t = title.strip()
if _REFERENCE_RE.match(t):
return True
if re.search(r'\(\*+\)', t): # marked spot "(**)" / "(*)" if re.search(r'\(\*+\)', t): # marked spot "(**)" / "(*)"
return True return True
return False return bool(_REFERENCE_RE.match(t)) or not _reference_strip(t)
def _canonical(candidates: list[dict], idxs: list[int], seen_norm: set[str]) -> dict: def _canonical(candidates: list[dict], idxs: list[int], seen_norm: set[str]) -> dict:
@@ -1118,6 +1028,15 @@ _CANON_STOP = re.compile(
r'von|of|für|for|und|and|zum|zur|im)\b', re.I) r'von|of|für|for|und|and|zum|zur|im)\b', re.I)
# German compounds are head-final: "Cliquenproblem" = stem + optional Fuge + head noun "problem".
# The standalone wrapper "problem" is already in _CANON_STOP, but a GLUED head hides the stem from ER
# blocking ("Cliquenproblem" would never share a key with "Clique"). Strip the glued "problem" head so
# the stem surfaces — a generic suffix rule, not a term list. The Fuge is only a consonant n/s
# ("Cliquen-", "Entscheidungs-"): matching a vowel "e" would eat the stem's own "e" ("clique" → "cliqu").
# The ≥3-char stem lookbehind keeps a card literally titled "Problem"/"Probleme" (→ _CANON_STOP) intact.
_CANON_GLUED_HEAD = re.compile(r'(?<=\w{3})[ns]?problem(?:en|e|s)?\b', re.I)
# Catalogue references ("Definition 6.19", "Satz 7.8") are scaffolding INCLUDING their number — # Catalogue references ("Definition 6.19", "Satz 7.8") are scaffolding INCLUDING their number —
# stripped as a phrase, so the digits don't pollute the key. Variant digits ("3-SAT") have no # stripped as a phrase, so the digits don't pollute the key. Variant digits ("3-SAT") have no
# scaffolding word in front and survive. # scaffolding word in front and survive.
@@ -1152,6 +1071,7 @@ def _canonical_key(title: str) -> str:
s = _CANON_CATALOGUE.sub(' ', s) s = _CANON_CATALOGUE.sub(' ', s)
s = re.sub(r'≟|\bversus\b|\bvs\.?\b|=', ' opeq ', s) # equality / "vs" → one token s = re.sub(r'≟|\bversus\b|\bvs\.?\b|=', ' opeq ', s) # equality / "vs" → one token
s = re.sub(r'≤|⪯|→|⇒|⟹|\breduces?\s+to\b|\breduziert\b', ' opred ', s) # reduction → one token s = re.sub(r'≤|⪯|→|⇒|⟹|\breduces?\s+to\b|\breduziert\b', ' opred ', s) # reduction → one token
s = _CANON_GLUED_HEAD.sub(' ', s) # glued "…problem" head → stem
s = _CANON_STOP.sub(' ', s) s = _CANON_STOP.sub(' ', s)
s = re.sub(r'[^\w ]', ' ', s) # drop punctuation/symbols s = re.sub(r'[^\w ]', ' ', s) # drop punctuation/symbols
return " ".join(sorted(t for t in s.split() if t)) return " ".join(sorted(t for t in s.split() if t))
@@ -1340,16 +1260,21 @@ _REDUCTION_ONLY_OP = re.compile(r'[≤⪯]|→|⇒|⟹|->') # genuine reduction
def _is_reduction_statement(title: str) -> bool: def _is_reduction_statement(title: str) -> bool:
"""True if the title is a reduction between two NAMED problems (both sides carry ≥3 letters and """True if the title is a reduction between two NAMED problems (both sides carry ≥3 letters and
neither is a pure bound like "O(m)"). Rejects unary "X ∈ NP" and proof-size "|A| = O(m)".""" neither is a pure bound like "O(m)"). Rejects unary "X ∈ NP", proof-size "|A| = O(m)", and a
construction/assignment suffix after the target ("3-SAT ≤ K-COLOR: G=(V,E) Konstruktion" is a
construction fragment, not a standalone statement — a bare target carries no ':' or '=')."""
t = _REL_STRIP.sub('', title, count=1) t = _REL_STRIP.sub('', title, count=1)
m = _REDUCTION_ONLY_OP.search(t) m = _REDUCTION_ONLY_OP.search(t)
if not m: if not m:
return False return False
right = t[m.end():]
if ':' in right or '=' in right: # construction/assignment detail after the target → fragment
return False
def _named(s: str) -> bool: def _named(s: str) -> bool:
return len(re.findall(r'[a-zäöüß]', s, re.I)) >= 3 and not re.match(r'\s*[Oo]\s*\(', s) return len(re.findall(r'[a-zäöüß]', s, re.I)) >= 3 and not re.match(r'\s*[Oo]\s*\(', s)
return _named(t[:m.start()]) and _named(t[m.end():]) return _named(t[:m.start()]) and _named(right)
def _is_named_statement(title: str, desc: str = "") -> bool: def _is_named_statement(title: str, desc: str = "") -> bool:

View File

@@ -43,7 +43,7 @@ from blocks import (
_build_research_prompt, _canonical, _canonical_key, _chunk_nums, _cliques, _build_research_prompt, _canonical, _canonical_key, _chunk_nums, _cliques,
_completion_schema, _containment_parent, _crawl_index, _file_payload, _completion_schema, _containment_parent, _crawl_index, _file_payload,
_filter_schema, _filter_suspect, _is_artifact, _is_named_statement, _filter_schema, _filter_suspect, _is_artifact, _is_named_statement,
_is_parentless_noise, _is_reference, _pairs_schema, _read, _is_parentless_noise, _is_reference, _reference_strip, _pairs_schema, _read,
_direction_conflict, _relation_conflict, _root, _supplement_schema, _text_sections, _umbrella_schema, _direction_conflict, _relation_conflict, _root, _supplement_schema, _text_sections, _umbrella_schema,
_aspect_marker, _title_variants, _corpus_files, _evidence_pack, _sink_json, source_folder, _aspect_marker, _title_variants, _corpus_files, _evidence_pack, _sink_json, source_folder,
) )
@@ -59,7 +59,7 @@ from fsutil import atomic_write_json, atomic_write_text
from jsonio import read_json_file as _json_file from jsonio import read_json_file as _json_file
from pipeline import (CANCELLED, FAILED, OK, GenContext, _extra, _log, _prompt, from pipeline import (CANCELLED, FAILED, OK, GenContext, _extra, _log, _prompt,
_runde_schema, _timeout, _yesno_schema, run_single_slot) _runde_schema, _timeout, _yesno_schema, run_single_slot)
from textkit import _norm_title, _parse_selection, _title, clean_title from textkit import _norm_title, _parse_selection, _title, _unclosed, clean_title
log = logging.getLogger("creator.board_inventory") log = logging.getLogger("creator.board_inventory")
@@ -491,12 +491,16 @@ def _hat_anker(title: str, ctoks: set[str]) -> bool:
def _sanierung_noetig(p: dict, ctoks: set[str] | None) -> bool: def _sanierung_noetig(p: dict, ctoks: set[str] | None) -> bool:
"""QA-messbare Befund-Formen am Entstehungsort: Titel ohne Korpus-Anker (QA: fremd — """QA-messbare Befund-Formen am Entstehungsort: Titel ohne Korpus-Anker (QA: fremd —
misst Token-Anker, nicht Semantik) oder leere Beschreibung (QA: hygiene). Gilt auch für misst Token-Anker, nicht Semantik) oder leere Beschreibung (QA: hygiene). Zusätzlich ein
Singleton-Cluster, die das Naming sonst überspringen — Reader-Rohtitel gingen wörtlich Referenz-Titel (reine Katalog-Nummer), eine unbalancierte Klammer oder ein überlanger Titel
bis done_block durch (gemessen: 'k-Coloring' statt Korpus-Form 'k-Color').""" (>80). Gilt auch für Singleton-Cluster, die das Naming sonst überspringen — Reader-Rohtitel
gingen wörtlich bis done_block durch (gemessen: 'k-Coloring' statt Korpus-Form 'k-Color')."""
title = p.get("title", "")
if not (p.get("description") or "").strip(): if not (p.get("description") or "").strip():
return True return True
return bool(ctoks) and not _hat_anker(p.get("title", ""), ctoks) if _is_reference(title) or _unclosed(title) or len(title) > 80:
return True
return bool(ctoks) and not _hat_anker(title, ctoks)
async def _anker_beleg(ctx: GenContext, flow: Flow, kandidaten: list[tuple[str, dict]]) -> set[str]: async def _anker_beleg(ctx: GenContext, flow: Flow, kandidaten: list[tuple[str, dict]]) -> set[str]:
@@ -574,6 +578,9 @@ async def _proc_consensus_gate(ctx: GenContext, flow: Flow, cards):
else: else:
moves.append((cid, "naming")) moves.append((cid, "naming"))
else: else:
# single-reader find: majority of the Klärung panel suffices (was unanimity — a single
# "nein" discarded a real once-mentioned concept as failed-quorum, e.g. Cook-Levin)
p["quorum"] = "majority"
moves.append((cid, "clarify")) moves.append((cid, "clarify"))
await db.kanban_set_payload(topic, BOARD, cid, p) await db.kanban_set_payload(topic, BOARD, cid, p)
if anker_kandidaten: if anker_kandidaten:
@@ -593,8 +600,9 @@ async def _proc_consensus_gate(ctx: GenContext, flow: Flow, cards):
async def _proc_clarify(ctx: GenContext, flow: Flow, cards): async def _proc_clarify(ctx: GenContext, flow: Flow, cards):
"""Single-reader finds: deterministic pre-reject, then a 3-judge panel (Blocks-Klaerung). """Single-reader finds: deterministic pre-reject, then a 3-judge panel (Blocks-Klaerung).
Quorum: UNANIMITY for single-reader clusters (origin-split C1), majority for reference-titled Quorum: MAJORITY of the panel for single-reader clusters and for reference-titled consensus
consensus clusters. Kept reference titles get the panel's rename.""" clusters (unanimity dropped real once-mentioned concepts on a single dissent). Kept reference
titles get the panel's rename."""
topic = flow.topic topic = flow.topic
moves: list[tuple[str, str]] = [] moves: list[tuple[str, str]] = []
pending = [] pending = []
@@ -810,6 +818,12 @@ async def _namecheck_one(ctx: GenContext, flow: Flow, c):
p["description"] = w.get("description") or p.get("description", "") p["description"] = w.get("description") or p.get("description", "")
readers = sorted(set().union(*[set(r.get("readers") or []) for r in rows])) if rows else [] readers = sorted(set().union(*[set(r.get("readers") or []) for r in rows])) if rows else []
sources = sorted(set().union(*[set(r.get("sources") or []) for r in rows])) if rows else [] sources = sorted(set().union(*[set(r.get("sources") or []) for r in rows])) if rows else []
# deterministic catalogue-number strip (final title): a source numbering is never canonical —
# "Satz 7.13 (Christofides)" → "Christofides", "Satz 7.6: Kriterium …" → "Kriterium …". Runs AFTER
# naming_check so a re-picked member title cannot re-introduce the number.
stripped = _reference_strip(p.get("title", ""))
if stripped and _norm_title(stripped) != _norm_title(p.get("title", "")) and not _is_reference(stripped):
p["title"] = clean_title(stripped)
await db.kanban_upsert_card(topic, BOARD, f"b-{cid}", "block", "fragment_filter", { await db.kanban_upsert_card(topic, BOARD, f"b-{cid}", "block", "fragment_filter", {
"title": p.get("title", ""), "description": p.get("description", ""), "title": p.get("title", ""), "description": p.get("description", ""),
"readers": readers, "sources": sources, "n_size": len(readers), "readers": readers, "sources": sources, "n_size": len(readers),
@@ -1025,13 +1039,21 @@ async def _proc_fragment_filter(ctx: GenContext, flow: Flow, cards):
for nr in nums: for nr in nums:
if nr in fragments or nr in honored: if nr in fragments or nr in honored:
continue continue
votes = len(dem.get(nr, []))
if drp.get(nr, 0) >= 2 and (_is_artifact(allrows[nr - 1]["title"]) if drp.get(nr, 0) >= 2 and (_is_artifact(allrows[nr - 1]["title"])
or _is_parentless_noise(allrows[nr - 1]["title"])): or _is_parentless_noise(allrows[nr - 1]["title"])):
honored.add(nr) honored.add(nr)
elif len(dem.get(nr, [])) >= 2:
fragments[nr] = max(set(dem[nr]), key=dem[nr].count)
elif nr in proposals: elif nr in proposals:
overruled.append(nr) # a pass-1 judge already proposed this demote — overturn it ONLY on a
# UNANIMOUS recheck keep (zero demote votes). 15/17 majority-rescued cards
# were later QA duplicates, so a single recheck demote reconfirms the fragment.
if votes >= 1:
fragments[nr] = max(set(dem[nr]), key=dem[nr].count)
else:
overruled.append(nr)
elif votes >= 2:
# pure ⚠ suspect without a pass-1 proposal: still needs a panel majority
fragments[nr] = max(set(dem[nr]), key=dem[nr].count)
# embedding backstop: veto confirmed non-containment demotes whose direct title pair is # embedding backstop: veto confirmed non-containment demotes whose direct title pair is
# literally structureless (see FRAGMENT_MIN_COS) — applied BEFORE _root resolution. # literally structureless (see FRAGMENT_MIN_COS) — applied BEFORE _root resolution.
floor_veto: list[int] = [] floor_veto: list[int] = []

View File

@@ -135,21 +135,6 @@ CONSENSUS_GRACE = 300
# check loops leave any remaining objections standing after that. # check loops leave any remaining objections standing after that.
CONSENSUS_MAX_ROUNDS = 3 CONSENSUS_MAX_ROUNDS = 3
# Crawler triage (content/noise) — deterministic rule filter instead of an LLM.
# Match: substring (lowercase) against URL AND file name. Order: keep > noise > min_chars > keep.
# Just add special rules here.
CRAWL_KEEP_PATTERNS = ["learn-unit", "learn-course"] # always content
CRAWL_NOISE_PATTERNS = [ # clearly off-topic → out
"clubs", "events", "podcasts", "resources", "-u-",
"academy", "pricing", "/plans", "career", "newsletter", "impressum", "login",
]
CRAWL_MIN_CHARS = 400 # too little text → out
# LLM topic relevance gate (after the rule filter): per content page yes/no against the spec.
# Separates the subject area (e.g. backend vs frontend), which the global CRAWL_* rules can't.
QUELLE_RELEVANZ_CHUNK = 12 # pages per rater package (small, since a snippet ships per page)
QUELLE_RELEVANZ_SNIPPET = 800 # body characters per page in the prompt (URL is the primary signal)
# QA gate: after the inventory phase an automatic QA run scores the blocks; below the # QA gate: after the inventory phase an automatic QA run scores the blocks; below the
# threshold the flow PAUSES before board 2 burns tokens (frontend offers force-continue). # threshold the flow PAUSES before board 2 burns tokens (frontend offers force-continue).
QA_GATE_NOTE = 9.5 # 0 = gate off; quota-based, so the tolerated finding count scales with topic size QA_GATE_NOTE = 9.5 # 0 = gate off; quota-based, so the tolerated finding count scales with topic size

View File

@@ -1,9 +1,9 @@
"""Bounded domain crawler for link sources — renders JS via Playwright (Chromium). """Page loader for link sources — renders JS via Playwright (Chromium).
Loads pages + PDFs starting from a start URL — ONLY the same domain, limited depth Loads a FIXED list of URLs the user supplied — one page each, NO link following.
and page count. HTML pages are rendered in a headless browser (needed for SPAs), then HTML pages are rendered in a headless browser (needed for SPAs), then the main text
links + text are pulled from the finished DOM. PDFs are loaded directly as bytes. is pulled from the finished DOM. PDFs are loaded directly as bytes. Deterministic,
Deterministic, bounded; runs via asyncio.to_thread (sync API, no event loop). bounded; runs via asyncio.to_thread (sync API, no event loop).
""" """
import hashlib import hashlib
@@ -17,8 +17,6 @@ from fsutil import atomic_write_text
log = logging.getLogger("creator.crawl") log = logging.getLogger("creator.crawl")
MAX_DEPTH = 3
MAX_PAGES = 500
PAGE_TIMEOUT = 30 # seconds per page (render or PDF download) PAGE_TIMEOUT = 30 # seconds per page (render or PDF download)
CRAWL_SETTLE_MS = 3000 # capped settle after domcontentloaded (SPA render); no 30s networkidle hang CRAWL_SETTLE_MS = 3000 # capped settle after domcontentloaded (SPA render); no 30s networkidle hang
MAX_BYTES = 10_000_000 # 10 MB cap per PDF MAX_BYTES = 10_000_000 # 10 MB cap per PDF
@@ -33,7 +31,7 @@ def _fetch_bytes(url: str) -> bytes | None:
data = resp.read(MAX_BYTES + 1) data = resp.read(MAX_BYTES + 1)
return None if len(data) > MAX_BYTES else data return None if len(data) > MAX_BYTES else data
except Exception as e: except Exception as e:
log.debug("crawl: PDF fetch failed %s: %s", url, e) log.debug("load: PDF fetch failed %s: %s", url, e)
return None return None
@@ -47,21 +45,6 @@ def _is_pdf(url: str) -> bool:
return url.lower().split("?")[0].rstrip("/").endswith(".pdf") return url.lower().split("?")[0].rstrip("/").endswith(".pdf")
def _scope_prefix(start_url: str) -> str:
"""First non-empty path segment of the start URL as the crawl scope, e.g.
`/learn/path/x` → `/learn`. No path segment → `""` (whole domain, no narrowing)."""
seg = [s for s in urlparse(start_url).path.split("/") if s]
return f"/{seg[0]}" if seg else ""
def _in_scope(url: str, prefix: str) -> bool:
"""Segment-exact prefix match (no `/learn` ⊃ `/learning-x`). Empty prefix → everything allowed."""
if not prefix:
return True
p = urlparse(url).path
return p == prefix or p.startswith(prefix + "/")
def _page_text(page) -> str: def _page_text(page) -> str:
"""Main text of the rendered page — nav/footer/boilerplate removed via trafilatura. """Main text of the rendered page — nav/footer/boilerplate removed via trafilatura.
Falls back to the raw body text when extraction is empty/too short (non-article pages).""" Falls back to the raw body text when extraction is empty/too short (non-article pages)."""
@@ -78,34 +61,34 @@ def _page_text(page) -> str:
return text.strip() return text.strip()
def crawl(start_url: str, target: Path, *, max_depth: int = MAX_DEPTH, max_pages: int = MAX_PAGES, cancelled=None) -> int: def load_pages(urls: list[str], target: Path, *, cancelled=None) -> int:
"""Crawl from start_url (same domain only), render JS and store pages/PDFs in `target`. """Load each URL in `urls` (one page each, NO link following), render JS and store
pages/PDFs in `target`.
BFS up to `max_depth` / `max_pages`. Errors on individual pages are skipped. Duplicate/fragment-only URLs collapse to one. Errors on individual pages are skipped.
Writes a `.done` marker at the END; an abort (`cancelled()` → True) omits it, Writes a `.done` marker at the END; an abort (`cancelled()` → True) omits it, so a
so a restart crawls again. Returns the number of saved sources. restart loads again. Returns the number of saved sources.
""" """
# Lazy: this way the backend starts even without Playwright installed; only crawling then fails. # Lazy: this way the backend starts even without Playwright installed; only loading then fails.
from playwright.sync_api import sync_playwright from playwright.sync_api import sync_playwright
target.mkdir(parents=True, exist_ok=True) target.mkdir(parents=True, exist_ok=True)
domain = urlparse(start_url).netloc
prefix = _scope_prefix(start_url) # only follow links under this path segment
seen: set[str] = set() seen: set[str] = set()
queue: list[tuple[str, int]] = [(urldefrag(start_url)[0], 0)] todo: list[str] = []
for u in urls:
nu = urldefrag(u)[0].strip()
if nu and nu not in seen:
seen.add(nu)
todo.append(nu)
saved = 0 saved = 0
with sync_playwright() as pw: with sync_playwright() as pw:
browser = pw.chromium.launch(args=["--no-sandbox"]) # non-root (Docker user app) browser = pw.chromium.launch(args=["--no-sandbox"]) # non-root (Docker user app)
page = browser.new_page(user_agent=_UA) page = browser.new_page(user_agent=_UA)
try: try:
while queue and saved < max_pages: for url in todo:
if cancelled and cancelled(): if cancelled and cancelled():
return saved # abort → NO .done marker → restart crawls again return saved # abort → NO .done marker → restart loads again
url, depth = queue.pop(0)
if url in seen:
continue
seen.add(url)
# PDFs need no rendering — load directly. # PDFs need no rendering — load directly.
if _is_pdf(url): if _is_pdf(url):
@@ -120,7 +103,7 @@ def crawl(start_url: str, target: Path, *, max_depth: int = MAX_DEPTH, max_pages
try: try:
page.goto(url, wait_until="domcontentloaded", timeout=PAGE_TIMEOUT * 1000) page.goto(url, wait_until="domcontentloaded", timeout=PAGE_TIMEOUT * 1000)
except Exception as e: except Exception as e:
log.debug("crawl: goto incomplete %s: %s", url, e) # still try to read the content log.debug("load: goto incomplete %s: %s", url, e) # still try to read the content
try: try:
page.wait_for_load_state("networkidle", timeout=CRAWL_SETTLE_MS) page.wait_for_load_state("networkidle", timeout=CRAWL_SETTLE_MS)
except Exception: except Exception:
@@ -129,20 +112,9 @@ def crawl(start_url: str, target: Path, *, max_depth: int = MAX_DEPTH, max_pages
if text: if text:
atomic_write_text(target / _name(url, ".txt"), f"QUELLE: {url}\n\n{text}") atomic_write_text(target / _name(url, ".txt"), f"QUELLE: {url}\n\n{text}")
saved += 1 saved += 1
if depth < max_depth:
try:
hrefs = page.eval_on_selector_all("a[href]", "els => els.map(e => e.href)")
except Exception:
hrefs = []
for href in hrefs:
nxt = urldefrag(href)[0]
if (nxt.startswith(("http://", "https://"))
and urlparse(nxt).netloc == domain and _in_scope(nxt, prefix)
and nxt not in seen):
queue.append((nxt, depth + 1))
finally: finally:
browser.close() browser.close()
(target / ".done").write_text("ok", encoding="utf-8") # ran through cleanly (target / ".done").write_text("ok", encoding="utf-8") # ran through cleanly
log.info("crawl %s%d sources in %s", start_url, saved, target) log.info("load_pages%d sources in %s", saved, target)
return saved return saved

View File

@@ -46,7 +46,7 @@ class BlocksCreateRequest(BaseModel):
instructions: str = Field(default="", max_length=2000) instructions: str = Field(default="", max_length=2000)
provider: ProviderType = DEFAULT_PROVIDER provider: ProviderType = DEFAULT_PROVIDER
source_type: SourceType = "thema" source_type: SourceType = "thema"
source_location: str = Field(default="", max_length=2000) source_location: str = Field(default="", max_length=20000) # link mode: one URL per line
research: bool = True # False = Continue: drain the existing kanban queue, no new search research: bool = True # False = Continue: drain the existing kanban queue, no new search
qa_force: bool = False # True = übersteuert ein pausierendes QA-Gate („Trotzdem fortsetzen") qa_force: bool = False # True = übersteuert ein pausierendes QA-Gate („Trotzdem fortsetzen")
@@ -97,7 +97,7 @@ class FolderResponse(BaseModel):
class BlocksSourceUpdate(BaseModel): class BlocksSourceUpdate(BaseModel):
topic: str = Field(min_length=1, max_length=100) topic: str = Field(min_length=1, max_length=100)
type: SourceType = "thema" type: SourceType = "thema"
location: str = Field(default="", max_length=2000) location: str = Field(default="", max_length=20000) # link mode: one URL per line
spec: str = Field(default="", max_length=2000) spec: str = Field(default="", max_length=2000)

View File

@@ -33,11 +33,14 @@ EMB_FLOOR = 0.82 # casefolded title cosine (own threshold, NOT the pipel
SECTION_CHARS = 4000 # own paragraph splitter — independent of _text_sections SECTION_CHARS = 4000 # own paragraph splitter — independent of _text_sections
COVER_MIN_TOKENS = 2 # distinctive block tokens a section must share to count as covered COVER_MIN_TOKENS = 2 # distinctive block tokens a section must share to count as covered
FREMD_MIN_TOKENS = 1 # distinctive title tokens that must appear in the corpus FREMD_MIN_TOKENS = 1 # distinctive title tokens that must appear in the corpus
CONCEPT_COVER_RATIO = 0.6 # fraction of a named result's stems a block must share to count as covered
LLM_SAMPLE = 12 # pairs/sections per judge call with --llm LLM_SAMPLE = 12 # pairs/sections per judge call with --llm
# Note 0-10, deterministisch aus den Quoten (transparent, diffbar — keine LLM-"Gefühlsnote"). # Note 0-10, deterministisch aus den Quoten (transparent, diffbar — keine LLM-"Gefühlsnote").
# Lücken/Fremd wiegen am schwersten (fehlender/falscher Stoff); Dubletten-VERDACHT enthält # Lücken/Fremd wiegen am schwersten (fehlender/falscher Stoff). konzept_luecken (benannte Kernresultate
# bewusst Rauschen und wiegt daher wenig. # ohne Baustein) ist heuristisch → mittleres Gewicht. Dubletten-VERDACHT enthält bewusst Rauschen,
NOTE_GEWICHTE = {"luecken": 3.0, "fremd": 2.5, "unechte_bloecke": 2.5, "hygiene": 0.5} # bildet aber das Nutzerproblem (Dopplungen) ab → leichtestes Gewicht, aber nicht null.
NOTE_GEWICHTE = {"luecken": 3.0, "fremd": 2.5, "unechte_bloecke": 2.5, "konzept_luecken": 1.5,
"dubletten_verdacht": 1.0, "hygiene": 0.5}
# subs/artefacts only exist after board 2 — at gate time these quotas would always be 0 # subs/artefacts only exist after board 2 — at gate time these quotas would always be 0
# and water down the inventory score, hence a separate score. # and water down the inventory score, hence a separate score.
# sub_dubletten counts only with --llm (confirmed pairs); the bare candidate list is # sub_dubletten counts only with --llm (confirmed pairs); the bare candidate list is
@@ -181,6 +184,48 @@ def luecken(blocks: list[dict], subs_by_norm: dict[str, list[str]], corpus: dict
return out return out
# A NAMED result carries a concept name, not just a number: "Satz 7.13 (Christofides)",
# "Satz 6.24: Cook-Levin". These are the core statements the token-based `luecken` misses — 108
# over-granular blocks cover every SECTION, yet the named core result may have no block at all.
# Generic + domain-safe: the NUMBER is mandatory (a numbered labeled unit is a formal statement in
# a structured document, any field), so prose like "ein Satz von Goethe" (no number) never matches.
_NAMED_RESULT_RE = re.compile(
r'\b(?:Satz|Lemma|Korollar|Theorem|Proposition|Folgerung|Definition|Algorithmus)\s+\d+(?:\.\d+)*\s*'
r'(?:\(\s*([^()\n]{3,60}?)\s*\)|:\s*([^\n.;·]{3,60}?)\s*(?:[.\n;·]|$))', re.M)
def _stem(t: str) -> str:
"""Declension-tolerant token stem: drop trailing digits, keep the 5-char prefix
('Eulerschen'/'Eulerscher''euler', 'Kreise'/'Kreis''kreis')."""
return t.rstrip("0123456789")[:5]
def _named_results(corpus: dict[str, str]) -> dict[str, set[str]]:
"""Named/attributed corpus results → {concept name: distinctive stems}. A bare 'Satz 7.18'
(number, no name) yields nothing to match. Same catalogue vocabulary as the title strip."""
out: dict[str, set[str]] = {}
for text in corpus.values():
for m in _NAMED_RESULT_RE.finditer(text):
name = (m.group(1) or m.group(2) or "").strip()
toks = _distinctive(name)
if len(name) >= 3 and toks:
out.setdefault(name, set()).update(_stem(t) for t in toks)
return out
def konzept_luecken(blocks: list[dict], named: dict[str, set[str]]) -> list[str]:
"""Named corpus results that NO block covers — concept gaps the token-based `luecken` cannot see.
Covered = a block whose title+description share ≥ CONCEPT_COVER_RATIO of the result's distinctive
stems (declension-tolerant). Errs toward 'covered' so the heuristic never invents a false gap."""
anchors = [{_stem(t) for t in _distinctive(b["title"]) | _distinctive(b.get("description") or "")}
for b in blocks]
out = []
for name, stems in named.items():
if not any(len(stems & a) >= CONCEPT_COVER_RATIO * len(stems) for a in anchors):
out.append(name)
return sorted(out)
def fremd(blocks: list[dict], corpus: dict[str, str]) -> list[str]: def fremd(blocks: list[dict], corpus: dict[str, str]) -> list[str]:
"""Blocks whose distinctive title tokens never appear in the corpus (scope creep). """Blocks whose distinctive title tokens never appear in the corpus (scope creep).
Token/stem match, NOT raw substring — 'bergang''Übergang' had whitewashed the Token/stem match, NOT raw substring — 'bergang''Übergang' had whitewashed the
@@ -384,6 +429,8 @@ async def qa_report(topic: str, llm: bool = False) -> dict | None:
sd = sub_dubletten(sub_rows) sd = sub_dubletten(sub_rows)
lk = luecken(blocks, subs_by_norm, corpus) if corpus else [] lk = luecken(blocks, subs_by_norm, corpus) if corpus else []
fr = fremd(blocks, corpus) if corpus else [] fr = fremd(blocks, corpus) if corpus else []
named = _named_results(corpus) if corpus else {}
kl = konzept_luecken(blocks, named) if corpus else []
bl = beleg(blocks, sub_rows) bl = beleg(blocks, sub_rows)
hy = hygiene(blocks) hy = hygiene(blocks)
n_sections = sum(len(_sections(t)) for t in corpus.values()) or 1 n_sections = sum(len(_sections(t)) for t in corpus.values()) or 1
@@ -452,13 +499,15 @@ async def qa_report(topic: str, llm: bool = False) -> dict | None:
"dubletten_verdacht": round(len(d) / max(len(blocks), 1), 3), "dubletten_verdacht": round(len(d) / max(len(blocks), 1), 3),
"luecken": round(len(_zaehlbare_luecken(lk, llm)) / n_sections, 3), "luecken": round(len(_zaehlbare_luecken(lk, llm)) / n_sections, 3),
"fremd": round(len(fr) / max(len(blocks), 1), 3), "fremd": round(len(fr) / max(len(blocks), 1), 3),
**({"konzept_luecken": round(len(kl) / max(len(named), 1), 3)} if corpus else {}),
"hygiene": round(len(hy) / max(len(blocks), 1), 3), "hygiene": round(len(hy) / max(len(blocks), 1), 3),
**({"unechte_bloecke": round(len(unecht) / max(len(blocks), 1), 3)} if unecht is not None else {}), **({"unechte_bloecke": round(len(unecht) / max(len(blocks), 1), 3)} if unecht is not None else {}),
}, },
"quoten_artefakte": quoten_art, "quoten_artefakte": quoten_art,
**({"unecht": unecht} if unecht is not None else {}), **({"unecht": unecht} if unecht is not None else {}),
**({"fremd_freigesprochen": fremd_frei} if fremd_frei else {}), **({"fremd_freigesprochen": fremd_frei} if fremd_frei else {}),
"dubletten": d, "sub_dubletten": sd, "luecken": lk, "fremd": fr, "beleg": bl, "hygiene": hy, "dubletten": d, "sub_dubletten": sd, "luecken": lk, "konzept_luecken": kl,
"fremd": fr, "beleg": bl, "hygiene": hy,
"artefakte": art, "artefakte": art,
"lauf": summary, "lauf": summary,
} }

View File

@@ -162,13 +162,7 @@ async def create_blocks(req: BlocksCreateRequest):
# Persist the source only the FIRST time; ▶/Resume keeps the existing choice. # Persist the source only the FIRST time; ▶/Resume keeps the existing choice.
if not qp.exists(): if not qp.exists():
type, location = req.source_type, req.source_location.strip() type, location = req.source_type, req.source_location.strip()
if type in ("projekt", "uni"): _validate_source(type, location)
folder = safe_folder(location)
if folder is None or not folder.is_dir():
raise HTTPException(400, "Folder invalid or not found (path relative to the project root, no ../).")
elif type == "link":
if not location.lower().startswith(("http://", "https://")):
raise HTTPException(400, "Link must start with http:// or https://.")
qp.parent.mkdir(parents=True, exist_ok=True) qp.parent.mkdir(parents=True, exist_ok=True)
atomic_write_json(qp, {"type": type, "location": location, "spec": req.instructions.strip()}) atomic_write_json(qp, {"type": type, "location": location, "spec": req.instructions.strip()})
asyncio.create_task(generate_blocks(topic, req.instructions.strip(), req.provider, asyncio.create_task(generate_blocks(topic, req.instructions.strip(), req.provider,
@@ -315,14 +309,17 @@ async def reset_block_progress(topic: str, block: str):
def _validate_source(type: str, location: str) -> None: def _validate_source(type: str, location: str) -> None:
"""Check source input (same rules as on creation).""" """Check source input (shared by create + edit)."""
if type in ("projekt", "uni"): if type in ("projekt", "uni"):
folder = safe_folder(location) folder = safe_folder(location)
if folder is None or not folder.is_dir(): if folder is None or not folder.is_dir():
raise HTTPException(400, "Folder invalid or not found (path relative to the project root, no ../).") raise HTTPException(400, "Folder invalid or not found (path relative to the project root, no ../).")
elif type == "link": elif type == "link":
if not location.lower().startswith(("http://", "https://")): urls = [ln.strip() for ln in location.splitlines() if ln.strip()]
raise HTTPException(400, "Link must start with http:// or https://.") if not urls:
raise HTTPException(400, "Enter at least one link.")
if not all(u.lower().startswith(("http://", "https://")) for u in urls):
raise HTTPException(400, "Each link must start with http:// or https:// (one per line).")
@router.get("/blocks/source", response_model=BlocksSourceResponse) @router.get("/blocks/source", response_model=BlocksSourceResponse)

View File

@@ -324,8 +324,9 @@ async def test_panel_confirms_demote(board_env, tmp_path, monkeypatch):
assert (await db.kanban_get_card(TOPIC, B, "b-2"))["stage"] == "dedup" assert (await db.kanban_get_card(TOPIC, B, "b-2"))["stage"] == "dedup"
async def test_panel_overrules_single_vote(board_env, tmp_path, monkeypatch): async def test_single_recheck_vote_confirms_proposal(board_env, tmp_path, monkeypatch):
"""Nur 1 von 3 Panel-Stimmen bestätigt den Judge-Demote → Karte überlebt (Journal).""" """M3.2: ein pass-1-Vorschlag wird schon von EINER Recheck-Stimme bestätigt → rejected.
(15/17 mehrheitlich geretteten Karten waren später QA-Dubletten — Overturn nur einstimmig.)"""
db, ctx, files = board_env db, ctx, files = board_env
monkeypatch.setattr(bi, "run_single_slot", _slot_router([ monkeypatch.setattr(bi, "run_single_slot", _slot_router([
("-filter-recheck-", _confirm_votes({"1"}, {"fragments": {"1": 2}, "drop": []})), ("-filter-recheck-", _confirm_votes({"1"}, {"fragments": {"1": 2}, "drop": []})),
@@ -335,6 +336,23 @@ async def test_panel_overrules_single_vote(board_env, tmp_path, monkeypatch):
("b-1", {"title": "Blockzitat", "description": "Zitat mit >"}), ("b-1", {"title": "Blockzitat", "description": "Zitat mit >"}),
("b-2", {"title": "Codeblock", "description": "Code mit Einrückung"}), ("b-2", {"title": "Codeblock", "description": "Code mit Einrückung"}),
]) ])
c1 = await db.kanban_get_card(TOPIC, B, "b-1")
assert c1["stage"] == "rejected"
assert c1["payload"]["reason"] == "fragment"
assert c1["payload"]["parent_norm"] == "codeblock"
async def test_unanimous_recheck_keep_overrules_proposal(board_env, tmp_path, monkeypatch):
"""M3.2: nur ein EINSTIMMIGES Recheck-Panel (0 Demote-Stimmen) hebt den pass-1-Vorschlag auf."""
db, ctx, files = board_env
monkeypatch.setattr(bi, "run_single_slot", _slot_router([
("-filter-recheck-", _confirm_votes(set(), {"fragments": {"1": 2}, "drop": []})),
("-filter-", {"fragments": {"1": 2}, "drop": []}),
]))
await _run_filter(db, ctx, tmp_path, [
("b-1", {"title": "Blockzitat", "description": "Zitat mit >"}),
("b-2", {"title": "Codeblock", "description": "Code mit Einrückung"}),
])
assert (await db.kanban_get_card(TOPIC, B, "b-1"))["stage"] == "dedup" assert (await db.kanban_get_card(TOPIC, B, "b-1"))["stage"] == "dedup"
journal = json.loads(next(tmp_path.glob("inventar-filter-*.json")).read_text(encoding="utf-8")) journal = json.loads(next(tmp_path.glob("inventar-filter-*.json")).read_text(encoding="utf-8"))
assert journal["ueberstimmt"] == ["Blockzitat"] assert journal["ueberstimmt"] == ["Blockzitat"]
@@ -587,6 +605,45 @@ def test_canonical_key_camel_and_catalogue():
assert k("Section 2.1 Matching") == k("Matching") assert k("Section 2.1 Matching") == k("Matching")
def test_canonical_key_glued_problem_suffix():
"""Verklebtes Kompositum-Suffix „…problem" wird abgetrennt, damit „Cliquenproblem" und
„Clique" denselben Blocking-Key teilen (aak: sonst nie Dedup-Kandidat). Fugen-n/-s + Plural
inklusive; die generische Suffix-Regel darf keine echten Varianten über-mergen."""
from blocks import _canonical_key as k
assert k("Clique") == k("Cliquenproblem") == k("Cliquenprobleme") != ""
assert k("Set Cover") == k("SetCover-Problem") # bestehender Hyphen-Pfad bleibt
assert k("SAT") != k("3-SAT") # Varianten-Ziffer bleibt Signal
assert k("Problem") != k("Clique") # bloßes „Problem" wird nicht zum Stamm
def test_reference_strip_and_is_reference():
"""Katalog-Nummern werden gestrippt, der Konzeptname bleibt; reine Nummern → leer + Referenz.
Reale aak-Schadensfälle (Satz-/Bemerkungs-Titel liefen wörtlich bis done)."""
from blocks import _reference_strip as strip, _is_reference as isref
assert strip("Satz 7.13 (Christofides)") == "Christofides"
assert strip("Satz 7.6: Kriterium für Eulerschen Kreis") == "Kriterium für Eulerschen Kreis"
assert strip("N P via nicht-deterministische Turingmaschine (Definition 6.19)") \
== "N P via nicht-deterministische Turingmaschine"
assert strip("Bemerkung 7.22") == ""
assert strip("Vertex Cover") == "Vertex Cover" # kein Katalog-Gerüst → unverändert
assert isref("Bemerkung 7.22") and isref("Satz 7.18") and isref("Korollar 6.18")
assert isref("Bedingung (**)")
assert not isref("Satz 7.13 (Christofides)") # hat Konzept → keine Referenz
assert not isref("Satz 7.6: Kriterium für Eulerschen Kreis")
assert not isref("P⊆NP") and not isref("Σ*") # kurze Symbole bleiben echt
def test_is_named_statement_construction_suffix():
"""Reduktion mit Konstruktions-Suffix (: / = nach dem Ziel) ist ein Fragment, kein Statement —
darf nicht mehr vor Demotion geschützt sein (aak: „3-SAT ≤ K-COLOR: G=(V,E) Konstruktion")."""
from blocks import _is_named_statement as named
assert not named("3-SAT ≤ K-COLOR: G=(V,E) Konstruktion")
assert named("3-SAT ≤ Clique") # saubere Reduktion bleibt geschützt
assert named("Clique → Vertex Cover")
assert not named("X ist NP-vollständig") # unäre Aussage bleibt demotable
assert named("Satz von Cook/Levin: SAT ist NP-vollständig ⇔ …") # benanntes Ergebnis bleibt
def test_relation_guard_ignores_trailing_scaffolding(): def test_relation_guard_ignores_trailing_scaffolding():
"""Trailing „Reduktion/Transformation" ist kein Operand — sonst blockt der Guard """Trailing „Reduktion/Transformation" ist kein Operand — sonst blockt der Guard
den korrekten Merge; Richtungs-Konflikte bleiben erkannt.""" den korrekten Merge; Richtungs-Konflikte bleiben erkannt."""
@@ -1213,6 +1270,49 @@ async def test_namecheck_ok_behaelt_titel(testdb, tmp_path, monkeypatch):
assert block["payload"]["description"] == "Eigene Beschreibung" assert block["payload"]["description"] == "Eigene Beschreibung"
async def test_namecheck_strips_source_numbering(testdb, tmp_path, monkeypatch):
"""M1.2: wählt der Check-Judge einen numerierten Titel, wird die Katalog-Nummer deterministisch
zum Konzept gestrippt ('Satz 7.13 (Christofides)''Christofides')."""
db = testdb
async def fake_members(topic, cid):
return [{"norm": "satz 7.13 (christofides)", "title": "Satz 7.13 (Christofides)",
"description": "3/2-Approximation für metrisches TSP", "readers": ["r1"], "sources": []},
{"norm": "b", "title": "B", "description": "db", "readers": ["r2"], "sources": []}]
async def fake_slot(ctx, label, *, key, prompt, role, capabilities, payload, timeout):
return "ok", payload((0, '{"best": 1}', "")) # Judge wählt den numerierten Member-Titel
monkeypatch.setattr(bi, "_member_rows", fake_members)
monkeypatch.setattr(bi, "run_single_slot", fake_slot)
payload = {"title": "Satz 7.13 (Christofides)", "description": "3/2-Approximation",
"main_norm": "satz 7.13 (christofides)"}
await db.kanban_upsert_card(TOPIC, B, "c7", "cluster", "naming_check", payload)
ctx = GenContext(topic=TOPIC, provider="claude", is_cancelled=lambda: False)
await bi._namecheck_one(ctx, _mk_flow(tmp_path), {"card_id": "c7", "payload": payload})
block = await db.kanban_get_card(TOPIC, B, "b-c7")
assert block["payload"]["title"] == "Christofides"
async def test_consensus_gate_single_reader_majority_quorum(testdb, tmp_path, monkeypatch):
"""M5.1: Einzel-Reader-Fund → clarify mit Mehrheits-Quorum statt Einstimmigkeit."""
db = testdb
monkeypatch.setattr(bi, "source_folder", lambda t: None) # thema: kein Korpus
async def fake_members(topic, cid):
return [{"title": "Seltenes Konzept", "description": "einmal erwähnt",
"readers": ["r1"], "supplement": False}]
monkeypatch.setattr(bi, "_member_rows", fake_members)
monkeypatch.setattr(bi, "_rep", lambda rows: rows[0])
await db.kanban_upsert_card(TOPIC, B, "cs", "cluster", "consensus_gate", {})
ctx = GenContext(topic=TOPIC, provider="claude", is_cancelled=lambda: False)
await bi._proc_consensus_gate(ctx, _mk_flow(tmp_path), [{"card_id": "cs", "payload": {}}])
card = await db.kanban_get_card(TOPIC, B, "cs")
assert card["stage"] == "clarify"
assert card["payload"]["quorum"] == "majority"
# ── Sanierung: Titel auf Korpus-Form, Beschreibungspflicht (QA: fremd/hygiene) ────── # ── Sanierung: Titel auf Korpus-Form, Beschreibungspflicht (QA: fremd/hygiene) ──────
def test_sanierung_schema_varianten(): def test_sanierung_schema_varianten():
@@ -1229,6 +1329,11 @@ def test_sanierung_noetig():
assert bi._sanierung_noetig({"title": "k-Coloring", "description": "d"}, ctoks) # kein Korpus-Anker assert bi._sanierung_noetig({"title": "k-Coloring", "description": "d"}, ctoks) # kein Korpus-Anker
assert not bi._sanierung_noetig({"title": "k-Color", "description": "d"}, ctoks) assert not bi._sanierung_noetig({"title": "k-Color", "description": "d"}, ctoks)
assert not bi._sanierung_noetig({"title": "k-Coloring", "description": "d"}, None) # thema: kein Korpus assert not bi._sanierung_noetig({"title": "k-Coloring", "description": "d"}, None) # thema: kein Korpus
# M1.3: Referenz-Titel / unbalancierte Klammer / >80 Zeichen triggern auch ohne Korpus (Singletons)
assert bi._sanierung_noetig({"title": "Bemerkung 7.22", "description": "d"}, None)
assert bi._sanierung_noetig({"title": "N P (Definition 6.19", "description": "d"}, None)
assert bi._sanierung_noetig({"title": "A" * 81, "description": "d"}, None)
assert not bi._sanierung_noetig({"title": "Clique", "description": "d"}, None)
def _sanierung_env(tmp_path, monkeypatch, antwort): def _sanierung_env(tmp_path, monkeypatch, antwort):

View File

@@ -93,9 +93,53 @@ def test_note_kalibrierung():
assert qa.note({"fremd": 1.0}) == 0.0 # komplett fremdes Inventar = 0, nicht 7.7 assert qa.note({"fremd": 1.0}) == 0.0 # komplett fremdes Inventar = 0, nicht 7.7
def test_note_verdacht_zaehlt_nicht(): def test_note_dubletten_verdacht_zaehlt():
"""dubletten_verdacht ist Verdachtsliste, kein Urteil — beeinflusst die Note nicht.""" """dubletten_verdacht bildet das Dopplungs-Nutzerproblem ab → zählt jetzt (Gewicht 1.0),
assert qa.note({"dubletten_verdacht": 1.0}) == 10.0 bleibt aber das leichteste Gewicht. 36 % Verdacht → 3.6 → 6.4; die Liste kann >1 sein → geklemmt."""
assert qa.note({"dubletten_verdacht": 0.0}) == 10.0
assert qa.note({"dubletten_verdacht": 0.36}) == 6.4
assert qa.note({"dubletten_verdacht": 2.0}) == 0.0 # >1 wird auf 1.0 geklemmt
# sub_dubletten_verdacht bleibt gewichtslos (Artefakt-Verdacht, kein Urteil)
assert qa.note({"sub_dubletten_verdacht": 1.0}, qa.NOTE_GEWICHTE_ARTEFAKTE) == 10.0
def test_note_konzept_luecken_zaehlt():
"""Fehlende benannte Kernresultate drücken die Note (Gewicht 1.5): 20 %3.0 → 7.0."""
assert qa.note({"konzept_luecken": 0.0}) == 10.0
assert qa.note({"konzept_luecken": 0.2}) == 7.0
def test_named_results_only_numbered(monkeypatch):
"""Benannte Ergebnisse: nur NUMMERIERTE Katalog-Referenzen mit Namen zählen — Prosa ohne
Nummer ('Satz von Goethe') und nackte Nummern ('Satz 7.18') liefern nichts (generisch)."""
corpus = {"skript.txt":
"Satz 7.13 (Christofides) liefert eine 3/2-Approximation.\n"
"Satz 6.24: Satz von Cook und Levin.\n"
"Nach Satz 7.18 folgt daraus die Schranke.\n"
"Ein Satz von Goethe steht hier."}
named = qa._named_results(corpus)
assert "Christofides" in named
assert any("Cook" in n and "Levin" in n for n in named)
assert not any("Goethe" in n for n in named) # keine Nummer → kein False Positive
assert all(n.strip() for n in named)
def test_konzept_luecken_flags_missing():
"""Benanntes Kernresultat ohne Baustein = Lücke; ein gedecktes Resultat nicht."""
corpus = {"s.txt": "Satz 7.13 (Christofides). Satz 6.16 (Kriterium für P=NP)."}
named = qa._named_results(corpus)
blocks = [{"title": "Christofides-Algorithmus", "description": "3/2-Approximation für TSP"}]
gaps = qa.konzept_luecken(blocks, named)
assert "Christofides" not in gaps
assert any("P=NP" in g or "Kriterium" in g for g in gaps)
def test_konzept_luecken_declension_tolerant():
"""Andere Flexion im Baustein deckt das Resultat trotzdem — kein falscher Lücken-Alarm."""
named = qa._named_results({"s.txt": "Satz 7.6: Kriterium für Eulerschen Kreis."})
blocks = [{"title": "Kriterium für Eulerscher Kreis", "description": "Grad aller Knoten gerade"}]
assert qa.konzept_luecken(blocks, named) == []
def test_note_artefakte_getrennt(): def test_note_artefakte_getrennt():

View File

@@ -0,0 +1,55 @@
"""Link source = curated URL list (one page per line, no crawl following)."""
from pathlib import Path
import pytest
from fastapi import HTTPException
import blocks
from pipeline import GenContext
from routes import _validate_source
# --- _validate_source: per-line validation ---------------------------------
def test_validate_source_link_multiline_ok():
_validate_source("link", "https://a.com/x\nhttp://b.com/y") # no raise
def test_validate_source_link_rejects_bad_line():
with pytest.raises(HTTPException):
_validate_source("link", "https://a.com/x\nftp://bad")
def test_validate_source_link_rejects_empty():
with pytest.raises(HTTPException):
_validate_source("link", " \n ")
# --- _prepare_source: load every line, all pages as content, no triage -----
async def test_prepare_source_link_loads_all_urls_as_content(testdb, tmp_path, monkeypatch):
topic = "T"
folder = tmp_path / "source"
folder.mkdir()
captured = {}
def fake_load(urls, target, *, cancelled=None):
captured["urls"] = list(urls)
for i, u in enumerate(urls):
(Path(target) / f"p{i}.txt").write_text(f"QUELLE: {u}\n\nInhalt {i}", encoding="utf-8")
return len(urls)
monkeypatch.setattr(blocks, "load_pages", fake_load)
monkeypatch.setattr(blocks, "_convert_pdfs", lambda f: None)
monkeypatch.setattr(blocks, "_crawl_done", lambda t: False)
monkeypatch.setattr(blocks, "_step_idx", lambda t, n: 0) # step list needs source.json — irrelevant here
ctx = GenContext(topic=topic, provider="claude", is_cancelled=lambda: False)
q = {"type": "link", "location": " https://a.com/x \n\nhttps://b.com/y\n", "spec": ""}
ok = await blocks._prepare_source(ctx, lambda *a, **k: None, {"arbeit": tmp_path}, q, folder, "")
assert ok is True
assert captured["urls"] == ["https://a.com/x", "https://b.com/y"] # split + strip + blanks dropped
content = await testdb.list_content(topic)
assert len(content) == 2 # every page kept, no relevance triage

View File

@@ -0,0 +1,57 @@
"""Titel-Hygiene: klammer-bewusster Split + Dash-Normalisierung (M1.1).
Reale aak-Schadensfälle: ein en-Dash mit einseitigem Space INNERHALB eines Titels
wurde zum Separator → Split mitten im Titel („Aε-Algorithmus (Güte 1+ε, Laufzeit O(n3").
"""
from textkit import _norm_dash, _title, _split_top, _unclosed, _parse_selection
def test_norm_dash_top_level_only():
# einseitiger Space am Dash → Separator (Modelle liefern das so)
assert _norm_dash("Titel Beschreibung") == "Titel — Beschreibung"
assert _norm_dash("Titel— Beschreibung") == "Titel — Beschreibung"
# Dash INNERHALB offener Klammer bleibt unangetastet (kein Separator)
assert _norm_dash("f (a b) — Rest") == "f (a b) — Rest"
# ASCII-Hyphen und geklebte/rangebasierte Dashes sind nie Separatoren
assert _norm_dash("3-SAT") == "3-SAT"
assert _norm_dash("1215") == "1215"
def test_split_top_bracket_aware():
assert _split_top("a (b — c) — d", "") == "a (b — c)"
assert _split_top("a — b", "") == "a"
assert _split_top("kein Separator hier", "") == "kein Separator hier"
def test_unclosed_bracket_detection():
assert _unclosed("a (b") is True
assert _unclosed("a [b {c") is True
assert _unclosed("a (b) c") is False
assert _unclosed("a) b") is False # überzähliger Schließer ist kein Abschnitt
def test_title_keeps_bracketed_dash():
"""Der Separator innerhalb der Klammer zerschneidet den Titel nicht mehr."""
t = _title("Aε-Algorithmus (Güte 1+ε Laufzeit O(n³)) — der Approx-Algorithmus")
assert t == "Aε-Algorithmus (Güte 1+ε Laufzeit O(n³))"
def test_title_plain_split_unchanged():
assert _title("Vertex Cover — minimale Knotenüberdeckung") == "Vertex Cover"
assert _title("Ohne Separator bleibt ganz") == "Ohne Separator bleibt ganz"
def test_title_unclosed_bracket_keeps_entry():
"""Malformed: offene Klammer nie geschlossen → keinen abgeschnittenen Titel ausgeben,
stattdessen den ganzen Eintrag behalten (Repair = Split rückgängig)."""
e = "N P via Turingmaschine (Definition 6.19 — Rest"
assert _title(e) == e
def test_parse_selection_then_title_regression():
"""End-to-end der Parse-Kette (wie in der Pipeline): _norm_dash + _title lassen den
Klammer-Inhalt intakt statt mitten im Titel zu splitten (der reale aak-Bug)."""
line = "1. Aε-Algorithmus (Güte 1+ε Laufzeit O(n³)) — der Approximations-Algorithmus"
entries = _parse_selection(line)
assert _title(entries[1]) == "Aε-Algorithmus (Güte 1+ε Laufzeit O(n³))"

View File

@@ -32,8 +32,46 @@ def _norm_title(s: str) -> str:
return s.casefold() return s.casefold()
# Bracket-aware separator handling: a ' — ' (or a dash that _norm_dash would normalize) INSIDE
# ()[]{} is part of the title, not a title/description separator. Without this a spaced dash within
# a math title truncates it: 'Aε-Algorithmus (Güte 1+ε, Laufzeit O(n³)) — …' → 'Aε-Algorithmus (Güte…'.
_BRACKETS = {"(": ")", "[": "]", "{": "}"}
_CLOSERS = {v: k for k, v in _BRACKETS.items()}
def _unclosed(s: str) -> bool:
"""True if s has a dangling opening bracket ()[]{} — the symptom of a mid-bracket cut. Extra
closers (more ')' than '(') do NOT count; only an unclosed opener signals a truncated title."""
depth = 0
for c in s:
if c in _BRACKETS:
depth += 1
elif c in _CLOSERS and depth > 0:
depth -= 1
return depth > 0
def _split_top(s: str, sep: str) -> str:
"""First bracket-depth-0 segment of s split on sep; whole s if sep never occurs at depth 0."""
depth = 0
for i, c in enumerate(s):
if c in _BRACKETS:
depth += 1
elif c in _CLOSERS and depth > 0:
depth -= 1
elif depth == 0 and s.startswith(sep, i):
return s[:i]
return s
def _title(entry: str) -> str: def _title(entry: str) -> str:
return entry.split("")[0].strip() or entry """Title = text before the first TOP-LEVEL ''. A separator inside ()[]{} does not split
(keeps math titles intact); if the extracted title has an unclosed bracket (malformed source),
keep the whole entry rather than emit a truncated title."""
title = _split_top(entry, "")
if _unclosed(title):
return entry.strip() or entry
return title.strip() or entry
def clean_title(s: str) -> str: def clean_title(s: str) -> str:
@@ -74,6 +112,9 @@ def _resolve_title(idx: dict[str, int], t: str) -> int | None:
return idx.get(_norm_title(t)) or idx.get(_norm_title(_title(t))) return idx.get(_norm_title(t)) or idx.get(_norm_title(_title(t)))
_DASH_SEP_RE = re.compile(r"\s*[‒–—―]\s+|\s+[‒–—―]\s*")
def _norm_dash(s: str) -> str: def _norm_dash(s: str) -> str:
"""Dash variants (en/em/figure/bar) with whitespace on AT LEAST ONE side → uniform separator ''. """Dash variants (en/em/figure/bar) with whitespace on AT LEAST ONE side → uniform separator ''.
Some models (especially non-western ones) use an en-dash "" instead of the em-dash; without Some models (especially non-western ones) use an en-dash "" instead of the em-dash; without
@@ -81,8 +122,12 @@ def _norm_dash(s: str) -> str:
space ("Titel —Beschreibung" / "Titel— Beschreibung") also breaks the split and leaks the source space ("Titel —Beschreibung" / "Titel— Beschreibung") also breaks the split and leaks the source
filename into the description — so a dash with a space on either side is repaired too. The ASCII filename into the description — so a dash with a space on either side is repaired too. The ASCII
hyphen "-" is deliberately NOT in the class (would split "n - 1"/"3-SAT"); requiring ≥1 surrounding hyphen "-" is deliberately NOT in the class (would split "n - 1"/"3-SAT"); requiring ≥1 surrounding
space keeps glued compounds like "Backtracking—Verfahren" and number ranges like "1215" untouched.""" space keeps glued compounds like "Backtracking—Verfahren" and number ranges like "1215" untouched.
return re.sub(r"\s*[‒–—―]\s+|\s+[‒–—―]\s*", "", s) A qualifying dash INSIDE an open bracket ()[]{} is left untouched — it is part of the title, not a
separator (else 'Algorithmus (Güte 1+ε O(n³)) — …' would split mid-parenthetical)."""
def _repl(m):
return m.group(0) if _unclosed(s[:m.start()]) else ""
return _DASH_SEP_RE.sub(_repl, s)
def _parse_selection(text: str) -> dict[int, str]: def _parse_selection(text: str) -> dict[int, str]:

View File

@@ -36,7 +36,6 @@ PARAMS: dict[str, dict] = {
"MAX_RESTARTS": {"default": 2, "min": 1, "max": 3, "step": 1, "kategorie": "laufzeit", "fidelity": "board2"}, "MAX_RESTARTS": {"default": 2, "min": 1, "max": 3, "step": 1, "kategorie": "laufzeit", "fidelity": "board2"},
"HEDGE_NACH_S": {"default": 90, "min": 30, "max": 240, "step": 30, "kategorie": "laufzeit", "fidelity": "board2"}, "HEDGE_NACH_S": {"default": 90, "min": 30, "max": 240, "step": 30, "kategorie": "laufzeit", "fidelity": "board2"},
"EVIDENCE_BUDGET_CHARS": {"default": 48000, "min": 16000, "max": 64000, "step": 8000, "kategorie": "tokens", "fidelity": "board2"}, "EVIDENCE_BUDGET_CHARS": {"default": 48000, "min": 16000, "max": 64000, "step": 8000, "kategorie": "tokens", "fidelity": "board2"},
"QUELLE_RELEVANZ_CHUNK": {"default": 12, "min": 6, "max": 24, "step": 3, "kategorie": "laufzeit", "fidelity": "voll"},
} }

View File

@@ -27,12 +27,12 @@ function setOrt(ort) {
<button v-for="t in TYPES" :key="t.key" :class="{ active: modelValue.type === t.key }" <button v-for="t in TYPES" :key="t.key" :class="{ active: modelValue.type === t.key }"
@click="setType(t.key)">{{ t.label }}</button> @click="setType(t.key)">{{ t.label }}</button>
</div> </div>
<input <textarea
v-if="modelValue.type === 'link'" v-if="modelValue.type === 'link'"
class="dlg-input" :value="modelValue.ort" class="dlg-input dlg-textarea" :value="modelValue.ort"
placeholder="https://…" rows="4"
placeholder="https://… — ein Link pro Zeile"
@input="setOrt($event.target.value)" @input="setOrt($event.target.value)"
@keyup.enter="emit('submit')"
/> />
<select <select
v-else-if="modelValue.type === 'projekt' || modelValue.type === 'uni'" v-else-if="modelValue.type === 'projekt' || modelValue.type === 'uni'"
@@ -69,4 +69,5 @@ function setOrt(ort) {
} }
.dlg-input:focus { outline: none; border-color: var(--accent); } .dlg-input:focus { outline: none; border-color: var(--accent); }
select.dlg-input { cursor: pointer; } select.dlg-input { cursor: pointer; }
.dlg-textarea { resize: vertical; min-height: 5rem; }
</style> </style>

View File

@@ -14,7 +14,7 @@ What is NOT a block (→ discard, EVEN if a description is attached):
- Pure references WITHOUT substantive description: "Satz 7.18" (nothing meaningful after the —). - Pure references WITHOUT substantive description: "Satz 7.18" (nothing meaningful after the —).
- Too vague or evidently fabricated. - Too vague or evidently fabricated.
**Reference title WITH real content → include AND rename:** If an entry has an uninformative title ("Satz 7.18", "Korollar 6.18", "Lemma 6.2") but the description describes a real concept → include it and give a speaking name (in GERMAN) drawn from the content in the `rename` field. In the `keep` entry the **original title** stays (for matching); the new name goes ONLY in `rename`. **Reference title WITH real content → include AND rename:** If an entry has an uninformative title ("Satz 7.18", "Korollar 6.18", "Lemma 6.2") but the description describes a real concept → include it and give a speaking name (in GERMAN) drawn from the content in the `rename` field. In the `keep` entry the **original title** stays (for matching); the new name goes ONLY in `rename`. A source numbering is never a canonical title — even with a name attached ("Satz 7.13 (Christofides)", "Satz 7.6: Kriterium …"), rename to the concept/result ("Satz von Christofides", "Kriterium für Eulerschen Kreis"); the numbering may remain in the description. A named theorem that states its own result stays a block — named after the RESULT, not the number.
Examples: Examples:
- "Clique-Member — k-Clique, die einen festen Knoten v enthält" → **keep** (standalone problem). - "Clique-Member — k-Clique, die einen festen Knoten v enthält" → **keep** (standalone problem).

View File

@@ -10,6 +10,7 @@ Check whether the current title is the best canonical name for this block.
Rules: Rules:
- Current title good (concrete, precise, self-explanatory, no reference/placeholder like "Satz 7.18") → confirm it. - Current title good (concrete, precise, self-explanatory, no reference/placeholder like "Satz 7.18") → confirm it.
- Otherwise pick the best member ("best" REQUIRED) and optionally propose a SHORTER name: max 8 words, ONLY terms that appear in the members, exactly as concrete as the shared content — never a broader umbrella term, never a textbook concept the members don't mention. NO catalog/reference brackets ("(Satz 6.33)"); a known acronym in brackets ("(VC)") is fine. - Otherwise pick the best member ("best" REQUIRED) and optionally propose a SHORTER name: max 8 words, ONLY terms that appear in the members, exactly as concrete as the shared content — never a broader umbrella term, never a textbook concept the members don't mention. NO catalog/reference brackets ("(Satz 6.33)"); a known acronym in brackets ("(VC)") is fine.
- A source numbering is never a canonical title, even with a name attached ("Satz 7.13 (Christofides)", "Satz 7.6: Kriterium …") → treat such a current title as NOT good and rename to the concept/result; the numbering belongs in the description.
- When in doubt, confirm the current title. - When in doubt, confirm the current title.
Reply with ONLY the JSON — no code fences, no other text. Reply with ONLY the JSON — no code fences, no other text.

View File

@@ -7,6 +7,7 @@ Rules:
- "best" is REQUIRED: the member whose title fits the shared concept best (most concrete, precise, self-explanatory, established term; avoid reference/placeholder titles like "Satz 7.18", "Punkt 3", "(**)"). - "best" is REQUIRED: the member whose title fits the shared concept best (most concrete, precise, self-explanatory, established term; avoid reference/placeholder titles like "Satz 7.18", "Punkt 3", "(**)").
- "name" is OPTIONAL — set it ONLY when no member title states the shared core well. It must be SHORT (max 8 words), use ONLY terms that appear in the members themselves, and stay exactly as concrete as the shared content — never a broader umbrella term, never a textbook concept the members don't mention. - "name" is OPTIONAL — set it ONLY when no member title states the shared core well. It must be SHORT (max 8 words), use ONLY terms that appear in the members themselves, and stay exactly as concrete as the shared content — never a broader umbrella term, never a textbook concept the members don't mention.
- In "name": NO catalog/reference brackets ("(Satz 6.33)", "(Kap. 4)"); a known acronym in brackets ("(VC)") is fine. - In "name": NO catalog/reference brackets ("(Satz 6.33)", "(Kap. 4)"); a known acronym in brackets ("(VC)") is fine.
- The title is the CONCEPT name — a source numbering is never canonical. Even with a name attached ("Satz 7.13 (Christofides)", "Satz 7.6: Kriterium für Eulerschen Kreis"), pick or propose the concept/result ("Satz von Christofides", "Kriterium für Eulerschen Kreis"); the numbering belongs in the description, never in the title.
Reply with ONLY the JSON — no code fences, no other text. Reply with ONLY the JSON — no code fences, no other text.

View File

@@ -8,9 +8,11 @@ Rules:
- Granularity: a block is a LEARNING UNIT, not a dictionary entry. Families learned together (e.g. the font-* properties, the modal verbs) are ONE block. - Granularity: a block is a LEARNING UNIT, not a dictionary entry. Families learned together (e.g. the font-* properties, the modal verbs) are ONE block.
- Abstraction level (important): a block is self-contained — explainable WITHOUT presupposing ANOTHER block as its subject. These are problems, methods/algorithms, definitions, named theorems with their own statement. A detail that presupposes another concept belongs IN that concept's block (as a part), not as its own entry: - Abstraction level (important): a block is self-contained — explainable WITHOUT presupposing ANOTHER block as its subject. These are problems, methods/algorithms, definitions, named theorems with their own statement. A detail that presupposes another concept belongs IN that concept's block (as a part), not as its own entry:
- A PROPERTY of a concept belongs to the concept: „Clique" is a block; „Clique ist NP-vollständig", „Clique ∈ NP", „Laufzeit von Clique" are parts of it — not their own entry. - A PROPERTY of a concept belongs to the concept: „Clique" is a block; „Clique ist NP-vollständig", „Clique ∈ NP", „Laufzeit von Clique" are parts of it — not their own entry.
- A PROOF/reduction component belongs to the theorem/reduction: „A-Komponente", „Dummy-Item", „αEnde", „Hilfsvariable" — not its own entry. - A PROOF/reduction component or a constructed HELPER OBJECT belongs to the theorem/reduction it serves: „A-Komponente", „Dummy-Item", „αEnde", „Hilfsvariable", helper functions/weightings, helper schedules, constructed graphs/tours/gadgets — not its own entry.
- A single COMPUTED instance or result value belongs to the argument that produced it: a specific number, graph, tour, or schedule built inside one example/proof („OPT(…) = n", a concrete constructed input) is not its own entry.
- PARAMETER INSTANCES of the SAME argument are ONE block: the identical technique applied to different parameters or cases („… via k / via n / via |E|", „für n=1 / n=2") is a single block, not one entry per parameter.
- Pure NOTATION/symbols belong to their definition: „|x|", „Σ∗" — not their own entry. - Pure NOTATION/symbols belong to their definition: „|x|", „Σ∗" — not their own entry.
- A standalone REDUCTION between two problems, however, is its own block („3-SAT ≤ Clique"). - A standalone REDUCTION between two problems, however, is its own block („3-SAT ≤ Clique"); likewise a named theorem that states its OWN result stays a block — named after the result, not the source numbering.
- NO categories, NO ranking, NO ordering by importance — only a flat, numbered list. - NO categories, NO ranking, NO ordering by importance — only a flat, numbered list.
- Aim for the natural number of genuine learning units for this material — there is no hard quota, but do NOT split hairs to inflate the count (prefer the "families learned together = ONE block" rule above). STOP when the only remaining candidates are exercises, meta-questions, administrative notes, or duplicates of blocks you already listed. - Aim for the natural number of genuine learning units for this material — there is no hard quota, but do NOT split hairs to inflate the count (prefer the "families learned together = ONE block" rule above). STOP when the only remaining candidates are exercises, meta-questions, administrative notes, or duplicates of blocks you already listed.
- **EXCLUDE non-content — these are NEVER learning blocks:** exercise/task/assignment scaffolding ("Aufgabe 3", "Bonusaufgabe", "Übung", "Blatt 11", point values), meta/assessment items ("welche der folgenden…", true/false prompts, hand-in / exam-date notes), administrative & boilerplate (headings, page/room/exam numbers, names, copyright), and any **file names, paths, or URLs** (those are provenance, recorded separately — never a block or part of one). - **EXCLUDE non-content — these are NEVER learning blocks:** exercise/task/assignment scaffolding ("Aufgabe 3", "Bonusaufgabe", "Übung", "Blatt 11", point values), meta/assessment items ("welche der folgenden…", true/false prompts, hand-in / exam-date notes), administrative & boilerplate (headings, page/room/exam numbers, names, copyright), and any **file names, paths, or URLs** (those are provenance, recorded separately — never a block or part of one).

View File

@@ -8,7 +8,7 @@ MATERIAL EXCERPTS:
{excerpts} {excerpts}
Tasks: Tasks:
- "title": If the title's terms do not literally appear in the excerpts, rewrite it using ONLY surface forms (exact spellings, even unusual ones) that appear in the excerpts — same meaning, same concreteness, max 8 words. If the title already matches the material wording, repeat it unchanged. Never broaden to a textbook term the excerpts don't use; NO catalog/reference brackets ("(Satz 6.33)", "(Kap. 4)"). - "title": If the title's terms do not literally appear in the excerpts, rewrite it using ONLY surface forms (exact spellings, even unusual ones) that appear in the excerpts — same meaning, same concreteness, max 8 words. If the title already matches the material wording, repeat it unchanged. Never broaden to a textbook term the excerpts don't use; NO catalog/reference brackets ("(Satz 6.33)", "(Kap. 4)"). A source numbering ("Satz 7.13", "Definition 6.19") is NOT a valid title even when it appears literally in the excerpts — use the concept/result name from the excerpts; the numbering may go into the description.
- "description": If the description is "(leer)", write ONE precise sentence grounded in the excerpts — no invented facts. Otherwise repeat it unchanged. - "description": If the description is "(leer)", write ONE precise sentence grounded in the excerpts — no invented facts. Otherwise repeat it unchanged.
Reply with ONLY the JSON — no code fences, no other text. Reply with ONLY the JSON — no code fences, no other text.

View File

@@ -1,23 +0,0 @@
You are reviewing crawled pages for the learning topic "{topic}". Decide per page: **ja** (belongs to the topic) or **nein** (off-topic). The selection controls which pages get read at all — off-topic pages bloat the inventory with foreign material.
LEARNING GOAL / SPEC (this is the yardstick):
{spec}
PAGES (number, then the URL, with a content excerpt below):
{pages}
How you decide:
- **ja**: The page covers exactly the subject area of the spec. Core, standard, or in-depth material on it. Even edge topics of the RIGHT field are `ja`.
- **nein**: A different subject area than the spec. Example: the spec is backend development, the page covers frontend, storefront, theming, payment operation, or an entirely different role/product area. Pure marketing, community, event, or account pages without learning material are also `nein`.
Important:
- Judge on the URL **and** content excerpt together. The URL slug is often the most telling.
- **Ignore the navigation menu / boilerplate** in the excerpt (login, „Sign up", menu items, footer) — it appears on every page and says nothing about the topic.
- **When in doubt, `ja`** — better to read one page too many than to exclude something important. Say `nein` only when the page clearly belongs to a different subject area.
- Evaluate each page individually. Invent nothing.
{extra}
Write ONLY the JSON file to: {out_path}
Format (exactly one value for EACH number; no other text):
{{"relevant": {{"1": "ja", "2": "nein", "3": "ja"}}}}