Files
codex-py/codex/wiki.py
Tarik Moussa 408e4886bb fix(wiki): basic conflict detection in CompileReport
Add _detect_conflicts() using adversative keyword heuristic (but,
however, in contrast, …) between chunks of different bibkeys; results
populate CompileReport.conflicts and appear in wiki/log.md.
Also add CompileReport.quarantined field (used by Fix 2).

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-14 05:11:06 +02:00

784 lines
26 KiB
Python

"""Wiki-compile layer — grounded concept pages over the RAG substrate (F-12).
Each concept page is compiled from retrieved chunks via a local LLM (Ollama).
Every claim is grounded against its cited source chunk (Substring-Match MVP).
Cross-references to other concepts are rendered as ``[[slug]]`` links.
Generated pages are written to ``wiki/<slug>.md`` and committed to git.
Compile-state (hash tracking for incremental re-runs) is persisted to
``wiki/.compile-state.json`` on disk — no DB changes (F-12 constraint).
Graceful degradation:
- Missing ``formulas`` table (F-09 not present) → no formula embedding, no crash.
- Missing ``verify_citations`` (F-10 not present) → local substring grounding check.
- LLM unavailable (httpx.ConnectError or any exception) → empty ConceptPage, no crash.
"""
from __future__ import annotations
import hashlib
import json
import logging
import re
import textwrap
from dataclasses import dataclass, field
from datetime import UTC, datetime
from pathlib import Path
from typing import Any, Protocol
import yaml
from codex.config import get_settings
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Dataclasses
# ---------------------------------------------------------------------------
@dataclass
class Concept:
"""A curated concept seed from ``wiki/concepts.yaml``."""
slug: str
title: str
aliases: list[str]
emphasis: str | None = None
@dataclass
class Claim:
"""A single factual claim extracted from a synthesised concept page."""
text: str
bibkey: str
locator: str # e.g. "page 9" | "eq.(9)" | "chunk 42"
grounded: bool = True # set by Grounding-Guard
@dataclass
class ConceptPage:
"""The compiled wiki page for one concept."""
concept: Concept
markdown: str # final rendered markdown
claims: list[Claim] = field(default_factory=list)
chunk_hash: str = "" # SHA-256 of concatenated source chunks
compiled_at: datetime = field(default_factory=lambda: datetime.now(UTC))
@dataclass
class CompileReport:
"""Summary of a compile run (appended to ``wiki/log.md``)."""
ran_at: datetime = field(default_factory=lambda: datetime.now(UTC))
compiled: list[str] = field(default_factory=list) # slugs written/updated
skipped: list[str] = field(default_factory=list) # slugs skipped (unchanged)
ungrounded: list[tuple[str, str]] = field(default_factory=list) # (slug, claim_text)
conflicts: list[tuple[str, str, str]] = field(default_factory=list) # (slug, bibkey1, bibkey2)
quarantined: list[str] = field(default_factory=list) # slugs written to wiki/draft/
# ---------------------------------------------------------------------------
# LLM protocol (injectable for tests)
# ---------------------------------------------------------------------------
class LLMClient(Protocol):
"""Minimal protocol for an LLM that can generate text."""
def generate(self, prompt: str, model: str) -> str:
"""Return generated text for *prompt* using *model*."""
...
# ---------------------------------------------------------------------------
# Default Ollama LLM client
# ---------------------------------------------------------------------------
class OllamaClient:
"""Thin HTTP wrapper around the Ollama ``/api/generate`` endpoint."""
def __init__(self, base_url: str) -> None:
self._base_url = base_url.rstrip("/")
def generate(self, prompt: str, model: str) -> str: # noqa: D102
import httpx
url = f"{self._base_url}/api/generate"
payload = {"model": model, "prompt": prompt, "stream": False}
response = httpx.post(url, json=payload, timeout=120.0)
response.raise_for_status()
data: dict[str, Any] = response.json()
return str(data.get("response", ""))
# ---------------------------------------------------------------------------
# YAML loader
# ---------------------------------------------------------------------------
def load_concepts(path: str) -> list[Concept]:
"""Parse ``wiki/concepts.yaml`` and return a list of :class:`Concept` objects.
Each entry must have ``slug``, ``title``, and ``aliases`` (list).
``emphasis`` is optional.
"""
raw = Path(path).read_text(encoding="utf-8")
data: dict[str, Any] = yaml.safe_load(raw)
concepts: list[Concept] = []
for entry in data.get("concepts", []):
concepts.append(
Concept(
slug=str(entry["slug"]),
title=str(entry["title"]),
aliases=[str(a) for a in entry.get("aliases", [])],
emphasis=entry.get("emphasis") or None,
)
)
return concepts
# ---------------------------------------------------------------------------
# Retrieval helpers
# ---------------------------------------------------------------------------
def _retrieve_chunks(
queries: list[str],
*,
top_k: int,
) -> list[dict[str, Any]]:
"""Retrieve chunks via hybrid search (dense + FTS) from the DB.
Returns a list of dicts with keys: ``id``, ``paper_id``, ``ord``,
``content``, ``bibkey``. Reference-list chunks are filtered out
(ADR-F12: bibliography fragments pollute top-K).
"""
from codex.db import get_conn
from codex.embed import get_embedder
embedder = get_embedder()
combined_query = " ".join(queries)
dense_vec = embedder.encode_dense([combined_query])[0].tolist()
sql = """
SELECT
c.id,
c.paper_id,
c.ord,
c.content,
p.bibkey,
c.embedding <-> %(emb)s::vector AS dist
FROM chunks c
JOIN papers p ON p.id = c.paper_id
WHERE c.embedding IS NOT NULL
AND p.bibkey IS NOT NULL
ORDER BY c.embedding <-> %(emb)s::vector
LIMIT %(top_k)s
"""
with get_conn() as conn:
rows = conn.execute(
sql,
{"emb": dense_vec, "top_k": top_k * 2}, # over-fetch before filtering
).fetchall()
# Filter reference-list chunks: skip chunks whose content looks like a bibliography
# (heuristic: > 60 % of lines match "^\[\d+\]" or "^[A-Z][a-z]+,?\s+[A-Z]\.").
ref_pattern = re.compile(r"^\s*(\[\d+\]|[A-Z][a-z]+,?\s+[A-Z]\.)", re.MULTILINE)
filtered: list[dict[str, Any]] = []
for row in rows:
content: str = row["content"]
lines = content.splitlines()
if not lines:
continue
ref_hits = len(ref_pattern.findall(content))
if ref_hits / max(len(lines), 1) > 0.6:
continue # skip reference-list chunk
filtered.append(dict(row))
if len(filtered) >= top_k:
break
return filtered
# ---------------------------------------------------------------------------
# Grounding guard
# ---------------------------------------------------------------------------
_CLAIM_RE = re.compile(
r"(?P<text>[^\[]+?)\s*\[(?P<bibkey>[^\],#]+)(?:#(?P<locator>[^\]]+))?\]",
)
def _parse_claims(markdown: str) -> list[Claim]:
"""Extract inline citations from the LLM output.
Expected format per claim::
Some factual statement. [BibKey2008 #page 9]
Returns a :class:`Claim` with ``text``, ``bibkey``, ``locator``.
The ``grounded`` flag defaults to ``True`` and is set by
:func:`_run_grounding_guard`.
"""
claims: list[Claim] = []
for match in _CLAIM_RE.finditer(markdown):
text = match.group("text").strip()
bibkey = match.group("bibkey").strip()
locator = (match.group("locator") or "").strip()
# Skip URL-shaped bibkeys ([text](https://...)) and multi-word bibkeys
# (real BibKeys never contain spaces or start with "http")
if " " in bibkey or bibkey.startswith("http"):
continue
if text and bibkey:
claims.append(Claim(text=text, bibkey=bibkey, locator=locator))
return claims
def _run_grounding_guard(
claims: list[Claim],
chunks: list[dict[str, Any]],
) -> list[Claim]:
"""Check each claim against its cited chunk via substring match (MVP).
A claim is *grounded* if at least one phrase from its text (> 4 words)
appears as a substring in a chunk attributed to the same bibkey,
OR if the claim text shares ≥ 3 consecutive words with any chunk of
that bibkey.
Sets ``claim.grounded = False`` for any claim that fails this check.
"""
# Build a bibkey → [content] index
bib_index: dict[str, list[str]] = {}
for chunk in chunks:
bk = str(chunk.get("bibkey") or "")
if bk:
bib_index.setdefault(bk, []).append(chunk["content"].lower())
for claim in claims:
sources = bib_index.get(claim.bibkey)
if not sources:
claim.grounded = False
continue
# Try to find any n-gram overlap (n ≥ 3 words)
words = claim.text.lower().split()
found = False
for n in range(min(len(words), 6), 2, -1): # try 6-grams down to 3-grams
for i in range(len(words) - n + 1):
phrase = " ".join(words[i : i + n])
if any(phrase in src for src in sources):
found = True
break
if found:
break
claim.grounded = found
return claims
# ---------------------------------------------------------------------------
# Conflict detection (MVP: keyword-based signal)
# ---------------------------------------------------------------------------
_CONFLICT_KEYWORDS = re.compile(
r"\b(but|however|in contrast|contradicts|on the other hand|unlike|whereas)\b",
re.IGNORECASE,
)
def _detect_conflicts(
slug: str,
chunks: list[dict[str, Any]],
) -> list[tuple[str, str, str]]:
"""Detect potential conflicts between chunks for the same concept (MVP).
Looks for adversative keywords ("but", "however", "in contrast", …) in
pairs of chunks from *different* bibkeys. Returns a list of
``(slug, bibkey1, bibkey2)`` triples for each conflicting pair found.
"""
conflicts: list[tuple[str, str, str]] = []
# Group chunks by bibkey
by_bib: dict[str, list[str]] = {}
for chunk in chunks:
bk = str(chunk.get("bibkey") or "")
if bk:
by_bib.setdefault(bk, []).append(chunk["content"])
bibkeys = list(by_bib.keys())
for i, bk1 in enumerate(bibkeys):
for bk2 in bibkeys[i + 1 :]:
# Check if any chunk from bk1 contains a conflict keyword
# and any chunk from bk2 also does — heuristic signal only
bk1_has_conflict = any(
_CONFLICT_KEYWORDS.search(c) for c in by_bib[bk1]
)
bk2_has_conflict = any(
_CONFLICT_KEYWORDS.search(c) for c in by_bib[bk2]
)
if bk1_has_conflict and bk2_has_conflict:
conflicts.append((slug, bk1, bk2))
return conflicts
# ---------------------------------------------------------------------------
# Cross-reference injection
# ---------------------------------------------------------------------------
_INLINE_CODE_RE = re.compile(r"`[^`]+`")
def _inject_cross_refs(
markdown: str,
all_concepts: list[Concept],
current_slug: str,
) -> str:
"""Replace occurrences of other concept titles/aliases with ``[[slug]]`` links.
Only exact case-insensitive whole-word matches outside of existing
``[[…]]`` blocks or inline code spans are replaced.
Inline-code spans (`` `…` ``) are temporarily protected by null-byte
placeholders and restored after injection.
"""
# Step 1: protect inline-code spans from replacement
placeholders: dict[str, str] = {}
def _protect(m: re.Match[str]) -> str:
key = f"\x00{len(placeholders)}\x00"
placeholders[key] = m.group(0)
return key
markdown = _INLINE_CODE_RE.sub(_protect, markdown)
# Step 2: inject cross-refs on unprotected text
for concept in all_concepts:
if concept.slug == current_slug:
continue
terms = [concept.title] + concept.aliases
for term in terms:
# Escape for use in regex; require word boundary
escaped = re.escape(term)
pattern = re.compile(rf"(?<!\[\[)\b{escaped}\b(?!\]\])", re.IGNORECASE)
replacement = f"[[{concept.slug}]]"
markdown = pattern.sub(replacement, markdown)
# Step 3: restore inline-code spans
for key, val in placeholders.items():
markdown = markdown.replace(key, val)
return markdown
# ---------------------------------------------------------------------------
# Chunk hash (for change detection)
# ---------------------------------------------------------------------------
def _chunk_hash(chunks: list[dict[str, Any]]) -> str:
"""Return a stable SHA-256 hex digest of the concatenated chunk contents."""
combined = "\n".join(c["content"] for c in sorted(chunks, key=lambda x: x["id"]))
return hashlib.sha256(combined.encode("utf-8")).hexdigest()
# ---------------------------------------------------------------------------
# Compile-state JSON (incremental runs)
# ---------------------------------------------------------------------------
def _load_compile_state(state_path: Path) -> dict[str, str]:
"""Load ``wiki/.compile-state.json`` → ``{slug: chunk_hash}`` dict."""
if not state_path.exists():
return {}
try:
raw = state_path.read_text(encoding="utf-8")
data: dict[str, str] = json.loads(raw)
return data
except (json.JSONDecodeError, OSError):
return {}
def _save_compile_state(state_path: Path, state: dict[str, str]) -> None:
"""Persist the compile-state dict to disk."""
state_path.write_text(json.dumps(state, indent=2, sort_keys=True), encoding="utf-8")
# ---------------------------------------------------------------------------
# LLM synthesis prompt
# ---------------------------------------------------------------------------
_SYNTHESIS_PROMPT_TEMPLATE = textwrap.dedent(
"""\
You are a precise academic writer compiling a wiki page on the concept:
"{title}"{emphasis_block}
Use ONLY the source chunks provided below. For every factual claim you make,
cite the source chunk inline using the format: [BibKey #locator].
Example: "The volume formula is V = L(γ₁)+L(γ₂)+L(γ₃). [Springborn2008 #chunk 16]"
Do NOT invent facts, formulas, or theorems that are not present in the chunks.
If a standard result is not in the chunks, do not include it.
Write 3-6 concise paragraphs. Use LaTeX math notation where appropriate ($ … $).
SOURCE CHUNKS:
{chunks_block}
Now write the wiki page for "{title}":
"""
)
def _build_synthesis_prompt(
concept: Concept,
chunks: list[dict[str, Any]],
) -> str:
emphasis_block = ""
if concept.emphasis:
emphasis_block = f"\n\nEmphasis: {concept.emphasis}"
chunks_block_lines = []
for chunk in chunks:
bibkey = chunk.get("bibkey") or chunk["paper_id"]
ord_val = chunk.get("ord", "?")
chunks_block_lines.append(f"[{bibkey} #chunk {ord_val}]\n{chunk['content'].strip()}\n")
chunks_block = "\n---\n".join(chunks_block_lines)
return _SYNTHESIS_PROMPT_TEMPLATE.format(
title=concept.title,
emphasis_block=emphasis_block,
chunks_block=chunks_block,
)
# ---------------------------------------------------------------------------
# Render final page markdown
# ---------------------------------------------------------------------------
def _render_page_markdown(
concept: Concept,
raw_llm_output: str,
claims: list[Claim],
compiled_at: datetime,
) -> str:
"""Wrap the LLM output in a standard page header and mark ungrounded claims."""
ungrounded_texts = {c.text for c in claims if not c.grounded}
body = raw_llm_output.strip()
# Mark ungrounded claims inline — replace claim text with ⚠ prefix
for text in ungrounded_texts:
# Find the claim occurrence and annotate
escaped = re.escape(text)
body = re.sub(
rf"({escaped})",
r"\1",
body,
count=1,
)
ts = compiled_at.strftime("%Y-%m-%d %H:%M UTC")
header = f"# {concept.title}\n\n_Compiled {ts} by `codex wiki compile`_\n\n"
return header + body + "\n"
# ---------------------------------------------------------------------------
# Core compile function
# ---------------------------------------------------------------------------
def compile_concept(
concept: Concept,
chunks: list[dict[str, Any]],
*,
top_k: int,
llm: LLMClient,
all_concepts: list[Concept] | None = None,
wiki_dir: Path | None = None,
) -> ConceptPage:
"""Compile a single concept page.
Parameters
----------
concept:
The concept to compile.
chunks:
Pre-retrieved source chunks for this concept (from :func:`_retrieve_chunks`).
Passing chunks explicitly avoids a second retrieve and ensures the stored
hash matches the chunks actually used for synthesis.
1. Synthesise via LLM (Ollama) with per-claim citation format.
2. Run Grounding-Guard: mark ungrounded claims as ⚠.
3. Inject cross-references to other concepts as [[slug]] links.
4. Embed formula chunks if ``formulas`` table is present (graceful).
Returns a :class:`ConceptPage` with full markdown and claim list.
"""
settings = get_settings()
_wiki_dir = wiki_dir or Path(settings.wiki_dir) # noqa: F841 — kept for future use
h = _chunk_hash(chunks)
prompt = _build_synthesis_prompt(concept, chunks)
try:
raw_output = llm.generate(prompt, model=settings.wiki_llm_model)
except Exception as exc: # noqa: BLE001
logger.warning("LLM unavailable (%s): skipping concept %s", exc, concept.slug)
return ConceptPage(concept=concept, markdown="", claims=[], chunk_hash=h)
claims = _parse_claims(raw_output)
claims = _run_grounding_guard(claims, chunks)
_all_concepts = all_concepts or []
raw_output = _inject_cross_refs(raw_output, _all_concepts, concept.slug)
# Graceful: try to embed formula chunks (F-09) — skip if table missing
_try_embed_formulas(concept, chunks)
compiled_at = datetime.now(UTC)
markdown = _render_page_markdown(concept, raw_output, claims, compiled_at)
return ConceptPage(
concept=concept,
markdown=markdown,
claims=claims,
chunk_hash=h,
compiled_at=compiled_at,
)
def _try_embed_formulas(concept: Concept, chunks: list[dict[str, Any]]) -> None:
"""Attempt to look up formula chunks for the concept — graceful no-op if F-09 absent."""
try:
from codex.db import get_conn
with get_conn() as conn:
# Check if formulas table exists
row = conn.execute(
"SELECT 1 FROM information_schema.tables WHERE table_name = 'formulas'"
).fetchone()
if row is None:
return # F-09 not present
# (Future: embed relevant raw_latex into the page)
except Exception: # noqa: BLE001
return # DB not reachable or other error — degrade gracefully
# ---------------------------------------------------------------------------
# compile_all
# ---------------------------------------------------------------------------
def compile_all(
*,
changed_only: bool = True,
top_k: int | None = None,
concept_filter: str | None = None,
output_dir: str | None = None,
llm: LLMClient | None = None,
) -> CompileReport:
"""Compile all (or changed) concept pages.
Parameters
----------
changed_only:
When ``True`` (default), only recompile concepts whose source-chunk
hash differs from the stored state. ``False`` forces full recompile.
top_k:
Override ``config.wiki_top_k``.
concept_filter:
If set, compile only this concept slug.
output_dir:
Override ``config.wiki_dir``.
llm:
Injectable LLM client (defaults to :class:`OllamaClient`).
"""
settings = get_settings()
wiki_dir = Path(output_dir or settings.wiki_dir)
wiki_dir.mkdir(parents=True, exist_ok=True)
k = top_k if top_k is not None else settings.wiki_top_k
concepts_path = wiki_dir / "concepts.yaml"
if not concepts_path.exists():
# Fall back to sibling concepts.yaml next to wiki/ dir
concepts_path = wiki_dir.parent / "wiki" / "concepts.yaml"
# Load the FULL concept list — used for cross-reference injection regardless of filter
all_concepts = load_concepts(str(concepts_path))
# Apply filter only to the set of concepts that will be (re-)compiled
compile_concepts = (
[c for c in all_concepts if c.slug == concept_filter] if concept_filter else all_concepts
)
state_path = wiki_dir / ".compile-state.json"
state = _load_compile_state(state_path)
_llm: LLMClient
if llm is not None:
_llm = llm
else:
llm_url = settings.wiki_llm_url or settings.ollama_base_url
_llm = OllamaClient(llm_url)
report = CompileReport()
for concept in compile_concepts:
# Retrieve chunks once — reuse for hash check and synthesis
queries = [concept.title] + concept.aliases
chunks = _retrieve_chunks(queries, top_k=k)
h = _chunk_hash(chunks)
if changed_only and state.get(concept.slug) == h:
report.skipped.append(concept.slug)
continue
page = compile_concept(
concept,
chunks,
top_k=k,
llm=_llm,
all_concepts=all_concepts, # always the full list for cross-refs
wiki_dir=wiki_dir,
)
# Collect ungrounded claims for the report
for claim in page.claims:
if not claim.grounded:
report.ungrounded.append((concept.slug, claim.text))
# Detect conflicts between chunks from different bibkeys
report.conflicts.extend(_detect_conflicts(concept.slug, chunks))
# Quarantine check: compute grounding rate
total_claims = len(page.claims)
grounded_claims = sum(1 for c in page.claims if c.grounded)
grounding_rate = grounded_claims / total_claims if total_claims > 0 else 0.0
if total_claims > 0 and grounding_rate < settings.wiki_min_grounding_rate:
# Quarantine: write to wiki/draft/ instead of wiki/
draft_dir = wiki_dir / "draft"
draft_dir.mkdir(parents=True, exist_ok=True)
page_path = draft_dir / f"{concept.slug}.md"
page_path.write_text(page.markdown, encoding="utf-8")
report.quarantined.append(concept.slug)
else:
# Write page to wiki/
page_path = wiki_dir / f"{concept.slug}.md"
page_path.write_text(page.markdown, encoding="utf-8")
state[concept.slug] = page.chunk_hash
report.compiled.append(concept.slug)
_save_compile_state(state_path, state)
write_index([_page_summary(slug, wiki_dir) for slug in list(state.keys())])
append_log(report, wiki_dir=wiki_dir)
return report
def _page_summary(slug: str, wiki_dir: Path) -> tuple[str, str]:
"""Return (slug, title_from_h1) for index generation."""
page_path = wiki_dir / f"{slug}.md"
title = slug
if page_path.exists():
first_line = page_path.read_text(encoding="utf-8").splitlines()[0]
if first_line.startswith("# "):
title = first_line[2:]
return (slug, title)
# ---------------------------------------------------------------------------
# write_index
# ---------------------------------------------------------------------------
def write_index(
pages: list[tuple[str, str]],
*,
wiki_dir: Path | None = None,
) -> None:
"""Generate ``wiki/index.md`` with [[links]] to all compiled concept pages.
Parameters
----------
pages:
List of ``(slug, title)`` tuples.
wiki_dir:
Path to the wiki directory (defaults to ``config.wiki_dir``).
"""
settings = get_settings()
_wiki_dir = wiki_dir or Path(settings.wiki_dir)
ts = datetime.now(UTC).strftime("%Y-%m-%d %H:%M UTC")
lines = [
"# Wiki Index",
"",
f"_Generated {ts} by `codex wiki compile`_",
"",
"## Concepts",
"",
]
for slug, title in sorted(pages, key=lambda x: x[0]):
lines.append(f"- [[{slug}]] — {title}")
lines.append("")
content = "\n".join(lines)
(_wiki_dir / "index.md").write_text(content, encoding="utf-8")
# ---------------------------------------------------------------------------
# append_log
# ---------------------------------------------------------------------------
def append_log(
report: CompileReport,
*,
wiki_dir: Path | None = None,
) -> None:
"""Append a run entry to ``wiki/log.md`` (never overwrites).
Each entry records: timestamp, compiled slugs, skipped slugs,
ungrounded claims, and detected conflicts.
"""
settings = get_settings()
_wiki_dir = wiki_dir or Path(settings.wiki_dir)
log_path = _wiki_dir / "log.md"
ts = report.ran_at.strftime("%Y-%m-%d %H:%M:%S UTC")
entry_lines = [
f"\n## Run {ts}",
"",
]
if report.compiled:
entry_lines.append(f"**Compiled:** {', '.join(report.compiled)}")
if report.skipped:
entry_lines.append(f"**Skipped (unchanged):** {', '.join(report.skipped)}")
if report.ungrounded:
entry_lines.append("")
entry_lines.append("**⚠ Ungrounded claims:**")
for slug, text in report.ungrounded:
entry_lines.append(f"- `{slug}`: {text[:120]}")
if report.quarantined:
entry_lines.append("")
entry_lines.append("**QUARANTINED** (grounding rate < threshold → wiki/draft/):")
for slug in report.quarantined:
entry_lines.append(f"- `{slug}`")
if report.conflicts:
entry_lines.append("")
entry_lines.append("**⚠ Conflicts detected:**")
for slug, bk1, bk2 in report.conflicts:
entry_lines.append(f"- `{slug}`: conflicting claims in {bk1} vs {bk2}")
entry_lines.append("")
entry = "\n".join(entry_lines)
# Append-only: open in append mode
with log_path.open("a", encoding="utf-8") as fh:
if log_path.stat().st_size == 0:
fh.write("# Wiki Compile Log\n")
fh.write(entry)