_inject_cross_refs temporarily replaces backtick spans with null-byte placeholders before injecting [[slug]] cross-refs, then restores them. This prevents concept titles inside `code` from being rewritten. Add tests for inline-code protection and full-list cross-ref injection. Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
718 lines
23 KiB
Python
718 lines
23 KiB
Python
"""Wiki-compile layer — grounded concept pages over the RAG substrate (F-12).
|
|
|
|
Each concept page is compiled from retrieved chunks via a local LLM (Ollama).
|
|
Every claim is grounded against its cited source chunk (Substring-Match MVP).
|
|
Cross-references to other concepts are rendered as ``[[slug]]`` links.
|
|
Generated pages are written to ``wiki/<slug>.md`` and committed to git.
|
|
|
|
Compile-state (hash tracking for incremental re-runs) is persisted to
|
|
``wiki/.compile-state.json`` on disk — no DB changes (F-12 constraint).
|
|
|
|
Graceful degradation:
|
|
- Missing ``formulas`` table (F-09 not present) → no formula embedding, no crash.
|
|
- Missing ``verify_citations`` (F-10 not present) → local substring grounding check.
|
|
- LLM unavailable (httpx.ConnectError or any exception) → empty ConceptPage, no crash.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import json
|
|
import logging
|
|
import re
|
|
import textwrap
|
|
from dataclasses import dataclass, field
|
|
from datetime import UTC, datetime
|
|
from pathlib import Path
|
|
from typing import Any, Protocol
|
|
|
|
import yaml
|
|
|
|
from codex.config import get_settings
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Dataclasses
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@dataclass
|
|
class Concept:
|
|
"""A curated concept seed from ``wiki/concepts.yaml``."""
|
|
|
|
slug: str
|
|
title: str
|
|
aliases: list[str]
|
|
emphasis: str | None = None
|
|
|
|
|
|
@dataclass
|
|
class Claim:
|
|
"""A single factual claim extracted from a synthesised concept page."""
|
|
|
|
text: str
|
|
bibkey: str
|
|
locator: str # e.g. "page 9" | "eq.(9)" | "chunk 42"
|
|
grounded: bool = True # set by Grounding-Guard
|
|
|
|
|
|
@dataclass
|
|
class ConceptPage:
|
|
"""The compiled wiki page for one concept."""
|
|
|
|
concept: Concept
|
|
markdown: str # final rendered markdown
|
|
claims: list[Claim] = field(default_factory=list)
|
|
chunk_hash: str = "" # SHA-256 of concatenated source chunks
|
|
compiled_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
|
|
|
|
@dataclass
|
|
class CompileReport:
|
|
"""Summary of a compile run (appended to ``wiki/log.md``)."""
|
|
|
|
ran_at: datetime = field(default_factory=lambda: datetime.now(UTC))
|
|
compiled: list[str] = field(default_factory=list) # slugs written/updated
|
|
skipped: list[str] = field(default_factory=list) # slugs skipped (unchanged)
|
|
ungrounded: list[tuple[str, str]] = field(default_factory=list) # (slug, claim_text)
|
|
conflicts: list[tuple[str, str, str]] = field(default_factory=list) # (slug, bibkey1, bibkey2)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# LLM protocol (injectable for tests)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class LLMClient(Protocol):
|
|
"""Minimal protocol for an LLM that can generate text."""
|
|
|
|
def generate(self, prompt: str, model: str) -> str:
|
|
"""Return generated text for *prompt* using *model*."""
|
|
...
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Default Ollama LLM client
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class OllamaClient:
|
|
"""Thin HTTP wrapper around the Ollama ``/api/generate`` endpoint."""
|
|
|
|
def __init__(self, base_url: str) -> None:
|
|
self._base_url = base_url.rstrip("/")
|
|
|
|
def generate(self, prompt: str, model: str) -> str: # noqa: D102
|
|
import httpx
|
|
|
|
url = f"{self._base_url}/api/generate"
|
|
payload = {"model": model, "prompt": prompt, "stream": False}
|
|
response = httpx.post(url, json=payload, timeout=120.0)
|
|
response.raise_for_status()
|
|
data: dict[str, Any] = response.json()
|
|
return str(data.get("response", ""))
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# YAML loader
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def load_concepts(path: str) -> list[Concept]:
|
|
"""Parse ``wiki/concepts.yaml`` and return a list of :class:`Concept` objects.
|
|
|
|
Each entry must have ``slug``, ``title``, and ``aliases`` (list).
|
|
``emphasis`` is optional.
|
|
"""
|
|
raw = Path(path).read_text(encoding="utf-8")
|
|
data: dict[str, Any] = yaml.safe_load(raw)
|
|
concepts: list[Concept] = []
|
|
for entry in data.get("concepts", []):
|
|
concepts.append(
|
|
Concept(
|
|
slug=str(entry["slug"]),
|
|
title=str(entry["title"]),
|
|
aliases=[str(a) for a in entry.get("aliases", [])],
|
|
emphasis=entry.get("emphasis") or None,
|
|
)
|
|
)
|
|
return concepts
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Retrieval helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _retrieve_chunks(
|
|
queries: list[str],
|
|
*,
|
|
top_k: int,
|
|
) -> list[dict[str, Any]]:
|
|
"""Retrieve chunks via hybrid search (dense + FTS) from the DB.
|
|
|
|
Returns a list of dicts with keys: ``id``, ``paper_id``, ``ord``,
|
|
``content``, ``bibkey``. Reference-list chunks are filtered out
|
|
(ADR-F12: bibliography fragments pollute top-K).
|
|
"""
|
|
from codex.db import get_conn
|
|
from codex.embed import get_embedder
|
|
|
|
embedder = get_embedder()
|
|
combined_query = " ".join(queries)
|
|
dense_vec = embedder.encode_dense([combined_query])[0].tolist()
|
|
|
|
sql = """
|
|
SELECT
|
|
c.id,
|
|
c.paper_id,
|
|
c.ord,
|
|
c.content,
|
|
p.bibkey,
|
|
c.embedding <-> %(emb)s::vector AS dist
|
|
FROM chunks c
|
|
JOIN papers p ON p.id = c.paper_id
|
|
WHERE c.embedding IS NOT NULL
|
|
AND p.bibkey IS NOT NULL
|
|
ORDER BY c.embedding <-> %(emb)s::vector
|
|
LIMIT %(top_k)s
|
|
"""
|
|
with get_conn() as conn:
|
|
rows = conn.execute(
|
|
sql,
|
|
{"emb": dense_vec, "top_k": top_k * 2}, # over-fetch before filtering
|
|
).fetchall()
|
|
|
|
# Filter reference-list chunks: skip chunks whose content looks like a bibliography
|
|
# (heuristic: > 60 % of lines match "^\[\d+\]" or "^[A-Z][a-z]+,?\s+[A-Z]\.").
|
|
ref_pattern = re.compile(r"^\s*(\[\d+\]|[A-Z][a-z]+,?\s+[A-Z]\.)", re.MULTILINE)
|
|
filtered: list[dict[str, Any]] = []
|
|
for row in rows:
|
|
content: str = row["content"]
|
|
lines = content.splitlines()
|
|
if not lines:
|
|
continue
|
|
ref_hits = len(ref_pattern.findall(content))
|
|
if ref_hits / max(len(lines), 1) > 0.6:
|
|
continue # skip reference-list chunk
|
|
filtered.append(dict(row))
|
|
if len(filtered) >= top_k:
|
|
break
|
|
|
|
return filtered
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Grounding guard
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_CLAIM_RE = re.compile(
|
|
r"(?P<text>[^\[]+?)\s*\[(?P<bibkey>[^\],#]+)(?:#(?P<locator>[^\]]+))?\]",
|
|
)
|
|
|
|
|
|
def _parse_claims(markdown: str) -> list[Claim]:
|
|
"""Extract inline citations from the LLM output.
|
|
|
|
Expected format per claim::
|
|
|
|
Some factual statement. [BibKey2008 #page 9]
|
|
|
|
Returns a :class:`Claim` with ``text``, ``bibkey``, ``locator``.
|
|
The ``grounded`` flag defaults to ``True`` and is set by
|
|
:func:`_run_grounding_guard`.
|
|
"""
|
|
claims: list[Claim] = []
|
|
for match in _CLAIM_RE.finditer(markdown):
|
|
text = match.group("text").strip()
|
|
bibkey = match.group("bibkey").strip()
|
|
locator = (match.group("locator") or "").strip()
|
|
# Skip URL-shaped bibkeys ([text](https://...)) and multi-word bibkeys
|
|
# (real BibKeys never contain spaces or start with "http")
|
|
if " " in bibkey or bibkey.startswith("http"):
|
|
continue
|
|
if text and bibkey:
|
|
claims.append(Claim(text=text, bibkey=bibkey, locator=locator))
|
|
return claims
|
|
|
|
|
|
def _run_grounding_guard(
|
|
claims: list[Claim],
|
|
chunks: list[dict[str, Any]],
|
|
) -> list[Claim]:
|
|
"""Check each claim against its cited chunk via substring match (MVP).
|
|
|
|
A claim is *grounded* if at least one phrase from its text (> 4 words)
|
|
appears as a substring in a chunk attributed to the same bibkey,
|
|
OR if the claim text shares ≥ 3 consecutive words with any chunk of
|
|
that bibkey.
|
|
|
|
Sets ``claim.grounded = False`` for any claim that fails this check.
|
|
"""
|
|
# Build a bibkey → [content] index
|
|
bib_index: dict[str, list[str]] = {}
|
|
for chunk in chunks:
|
|
bk = str(chunk.get("bibkey") or "")
|
|
if bk:
|
|
bib_index.setdefault(bk, []).append(chunk["content"].lower())
|
|
|
|
for claim in claims:
|
|
sources = bib_index.get(claim.bibkey)
|
|
if not sources:
|
|
claim.grounded = False
|
|
continue
|
|
|
|
# Try to find any n-gram overlap (n ≥ 3 words)
|
|
words = claim.text.lower().split()
|
|
found = False
|
|
for n in range(min(len(words), 6), 2, -1): # try 6-grams down to 3-grams
|
|
for i in range(len(words) - n + 1):
|
|
phrase = " ".join(words[i : i + n])
|
|
if any(phrase in src for src in sources):
|
|
found = True
|
|
break
|
|
if found:
|
|
break
|
|
claim.grounded = found
|
|
|
|
return claims
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Cross-reference injection
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
_INLINE_CODE_RE = re.compile(r"`[^`]+`")
|
|
|
|
|
|
def _inject_cross_refs(
|
|
markdown: str,
|
|
all_concepts: list[Concept],
|
|
current_slug: str,
|
|
) -> str:
|
|
"""Replace occurrences of other concept titles/aliases with ``[[slug]]`` links.
|
|
|
|
Only exact case-insensitive whole-word matches outside of existing
|
|
``[[…]]`` blocks or inline code spans are replaced.
|
|
Inline-code spans (`` `…` ``) are temporarily protected by null-byte
|
|
placeholders and restored after injection.
|
|
"""
|
|
# Step 1: protect inline-code spans from replacement
|
|
placeholders: dict[str, str] = {}
|
|
|
|
def _protect(m: re.Match[str]) -> str:
|
|
key = f"\x00{len(placeholders)}\x00"
|
|
placeholders[key] = m.group(0)
|
|
return key
|
|
|
|
markdown = _INLINE_CODE_RE.sub(_protect, markdown)
|
|
|
|
# Step 2: inject cross-refs on unprotected text
|
|
for concept in all_concepts:
|
|
if concept.slug == current_slug:
|
|
continue
|
|
terms = [concept.title] + concept.aliases
|
|
for term in terms:
|
|
# Escape for use in regex; require word boundary
|
|
escaped = re.escape(term)
|
|
pattern = re.compile(rf"(?<!\[\[)\b{escaped}\b(?!\]\])", re.IGNORECASE)
|
|
replacement = f"[[{concept.slug}]]"
|
|
markdown = pattern.sub(replacement, markdown)
|
|
|
|
# Step 3: restore inline-code spans
|
|
for key, val in placeholders.items():
|
|
markdown = markdown.replace(key, val)
|
|
|
|
return markdown
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Chunk hash (for change detection)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _chunk_hash(chunks: list[dict[str, Any]]) -> str:
|
|
"""Return a stable SHA-256 hex digest of the concatenated chunk contents."""
|
|
combined = "\n".join(c["content"] for c in sorted(chunks, key=lambda x: x["id"]))
|
|
return hashlib.sha256(combined.encode("utf-8")).hexdigest()
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Compile-state JSON (incremental runs)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _load_compile_state(state_path: Path) -> dict[str, str]:
|
|
"""Load ``wiki/.compile-state.json`` → ``{slug: chunk_hash}`` dict."""
|
|
if not state_path.exists():
|
|
return {}
|
|
try:
|
|
raw = state_path.read_text(encoding="utf-8")
|
|
data: dict[str, str] = json.loads(raw)
|
|
return data
|
|
except (json.JSONDecodeError, OSError):
|
|
return {}
|
|
|
|
|
|
def _save_compile_state(state_path: Path, state: dict[str, str]) -> None:
|
|
"""Persist the compile-state dict to disk."""
|
|
state_path.write_text(json.dumps(state, indent=2, sort_keys=True), encoding="utf-8")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# LLM synthesis prompt
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_SYNTHESIS_PROMPT_TEMPLATE = textwrap.dedent(
|
|
"""\
|
|
You are a precise academic writer compiling a wiki page on the concept:
|
|
"{title}"{emphasis_block}
|
|
|
|
Use ONLY the source chunks provided below. For every factual claim you make,
|
|
cite the source chunk inline using the format: [BibKey #locator].
|
|
Example: "The volume formula is V = L(γ₁)+L(γ₂)+L(γ₃). [Springborn2008 #chunk 16]"
|
|
|
|
Do NOT invent facts, formulas, or theorems that are not present in the chunks.
|
|
If a standard result is not in the chunks, do not include it.
|
|
Write 3-6 concise paragraphs. Use LaTeX math notation where appropriate ($ … $).
|
|
|
|
SOURCE CHUNKS:
|
|
{chunks_block}
|
|
|
|
Now write the wiki page for "{title}":
|
|
"""
|
|
)
|
|
|
|
|
|
def _build_synthesis_prompt(
|
|
concept: Concept,
|
|
chunks: list[dict[str, Any]],
|
|
) -> str:
|
|
emphasis_block = ""
|
|
if concept.emphasis:
|
|
emphasis_block = f"\n\nEmphasis: {concept.emphasis}"
|
|
|
|
chunks_block_lines = []
|
|
for chunk in chunks:
|
|
bibkey = chunk.get("bibkey") or chunk["paper_id"]
|
|
ord_val = chunk.get("ord", "?")
|
|
chunks_block_lines.append(f"[{bibkey} #chunk {ord_val}]\n{chunk['content'].strip()}\n")
|
|
chunks_block = "\n---\n".join(chunks_block_lines)
|
|
|
|
return _SYNTHESIS_PROMPT_TEMPLATE.format(
|
|
title=concept.title,
|
|
emphasis_block=emphasis_block,
|
|
chunks_block=chunks_block,
|
|
)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Render final page markdown
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _render_page_markdown(
|
|
concept: Concept,
|
|
raw_llm_output: str,
|
|
claims: list[Claim],
|
|
compiled_at: datetime,
|
|
) -> str:
|
|
"""Wrap the LLM output in a standard page header and mark ungrounded claims."""
|
|
ungrounded_texts = {c.text for c in claims if not c.grounded}
|
|
|
|
body = raw_llm_output.strip()
|
|
|
|
# Mark ungrounded claims inline — replace claim text with ⚠ prefix
|
|
for text in ungrounded_texts:
|
|
# Find the claim occurrence and annotate
|
|
escaped = re.escape(text)
|
|
body = re.sub(
|
|
rf"({escaped})",
|
|
r"⚠ \1",
|
|
body,
|
|
count=1,
|
|
)
|
|
|
|
ts = compiled_at.strftime("%Y-%m-%d %H:%M UTC")
|
|
header = f"# {concept.title}\n\n_Compiled {ts} by `codex wiki compile`_\n\n"
|
|
return header + body + "\n"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Core compile function
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def compile_concept(
|
|
concept: Concept,
|
|
chunks: list[dict[str, Any]],
|
|
*,
|
|
top_k: int,
|
|
llm: LLMClient,
|
|
all_concepts: list[Concept] | None = None,
|
|
wiki_dir: Path | None = None,
|
|
) -> ConceptPage:
|
|
"""Compile a single concept page.
|
|
|
|
Parameters
|
|
----------
|
|
concept:
|
|
The concept to compile.
|
|
chunks:
|
|
Pre-retrieved source chunks for this concept (from :func:`_retrieve_chunks`).
|
|
Passing chunks explicitly avoids a second retrieve and ensures the stored
|
|
hash matches the chunks actually used for synthesis.
|
|
|
|
1. Synthesise via LLM (Ollama) with per-claim citation format.
|
|
2. Run Grounding-Guard: mark ungrounded claims as ⚠.
|
|
3. Inject cross-references to other concepts as [[slug]] links.
|
|
4. Embed formula chunks if ``formulas`` table is present (graceful).
|
|
|
|
Returns a :class:`ConceptPage` with full markdown and claim list.
|
|
"""
|
|
settings = get_settings()
|
|
_wiki_dir = wiki_dir or Path(settings.wiki_dir) # noqa: F841 — kept for future use
|
|
|
|
h = _chunk_hash(chunks)
|
|
prompt = _build_synthesis_prompt(concept, chunks)
|
|
try:
|
|
raw_output = llm.generate(prompt, model=settings.wiki_llm_model)
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.warning("LLM unavailable (%s): skipping concept %s", exc, concept.slug)
|
|
return ConceptPage(concept=concept, markdown="", claims=[], chunk_hash=h)
|
|
|
|
claims = _parse_claims(raw_output)
|
|
claims = _run_grounding_guard(claims, chunks)
|
|
|
|
_all_concepts = all_concepts or []
|
|
raw_output = _inject_cross_refs(raw_output, _all_concepts, concept.slug)
|
|
|
|
# Graceful: try to embed formula chunks (F-09) — skip if table missing
|
|
_try_embed_formulas(concept, chunks)
|
|
|
|
compiled_at = datetime.now(UTC)
|
|
markdown = _render_page_markdown(concept, raw_output, claims, compiled_at)
|
|
|
|
return ConceptPage(
|
|
concept=concept,
|
|
markdown=markdown,
|
|
claims=claims,
|
|
chunk_hash=h,
|
|
compiled_at=compiled_at,
|
|
)
|
|
|
|
|
|
def _try_embed_formulas(concept: Concept, chunks: list[dict[str, Any]]) -> None:
|
|
"""Attempt to look up formula chunks for the concept — graceful no-op if F-09 absent."""
|
|
try:
|
|
from codex.db import get_conn
|
|
|
|
with get_conn() as conn:
|
|
# Check if formulas table exists
|
|
row = conn.execute(
|
|
"SELECT 1 FROM information_schema.tables WHERE table_name = 'formulas'"
|
|
).fetchone()
|
|
if row is None:
|
|
return # F-09 not present
|
|
# (Future: embed relevant raw_latex into the page)
|
|
except Exception: # noqa: BLE001
|
|
return # DB not reachable or other error — degrade gracefully
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# compile_all
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def compile_all(
|
|
*,
|
|
changed_only: bool = True,
|
|
top_k: int | None = None,
|
|
concept_filter: str | None = None,
|
|
output_dir: str | None = None,
|
|
llm: LLMClient | None = None,
|
|
) -> CompileReport:
|
|
"""Compile all (or changed) concept pages.
|
|
|
|
Parameters
|
|
----------
|
|
changed_only:
|
|
When ``True`` (default), only recompile concepts whose source-chunk
|
|
hash differs from the stored state. ``False`` forces full recompile.
|
|
top_k:
|
|
Override ``config.wiki_top_k``.
|
|
concept_filter:
|
|
If set, compile only this concept slug.
|
|
output_dir:
|
|
Override ``config.wiki_dir``.
|
|
llm:
|
|
Injectable LLM client (defaults to :class:`OllamaClient`).
|
|
"""
|
|
settings = get_settings()
|
|
wiki_dir = Path(output_dir or settings.wiki_dir)
|
|
wiki_dir.mkdir(parents=True, exist_ok=True)
|
|
k = top_k if top_k is not None else settings.wiki_top_k
|
|
|
|
concepts_path = wiki_dir / "concepts.yaml"
|
|
if not concepts_path.exists():
|
|
# Fall back to sibling concepts.yaml next to wiki/ dir
|
|
concepts_path = wiki_dir.parent / "wiki" / "concepts.yaml"
|
|
# Load the FULL concept list — used for cross-reference injection regardless of filter
|
|
all_concepts = load_concepts(str(concepts_path))
|
|
|
|
# Apply filter only to the set of concepts that will be (re-)compiled
|
|
compile_concepts = (
|
|
[c for c in all_concepts if c.slug == concept_filter] if concept_filter else all_concepts
|
|
)
|
|
|
|
state_path = wiki_dir / ".compile-state.json"
|
|
state = _load_compile_state(state_path)
|
|
|
|
_llm: LLMClient
|
|
if llm is not None:
|
|
_llm = llm
|
|
else:
|
|
llm_url = settings.wiki_llm_url or settings.ollama_base_url
|
|
_llm = OllamaClient(llm_url)
|
|
|
|
report = CompileReport()
|
|
|
|
for concept in compile_concepts:
|
|
# Retrieve chunks once — reuse for hash check and synthesis
|
|
queries = [concept.title] + concept.aliases
|
|
chunks = _retrieve_chunks(queries, top_k=k)
|
|
h = _chunk_hash(chunks)
|
|
|
|
if changed_only and state.get(concept.slug) == h:
|
|
report.skipped.append(concept.slug)
|
|
continue
|
|
|
|
page = compile_concept(
|
|
concept,
|
|
chunks,
|
|
top_k=k,
|
|
llm=_llm,
|
|
all_concepts=all_concepts, # always the full list for cross-refs
|
|
wiki_dir=wiki_dir,
|
|
)
|
|
|
|
# Write page to disk
|
|
page_path = wiki_dir / f"{concept.slug}.md"
|
|
page_path.write_text(page.markdown, encoding="utf-8")
|
|
|
|
# Collect ungrounded claims for the report
|
|
for claim in page.claims:
|
|
if not claim.grounded:
|
|
report.ungrounded.append((concept.slug, claim.text))
|
|
|
|
state[concept.slug] = page.chunk_hash
|
|
report.compiled.append(concept.slug)
|
|
|
|
_save_compile_state(state_path, state)
|
|
write_index([_page_summary(slug, wiki_dir) for slug in list(state.keys())])
|
|
append_log(report, wiki_dir=wiki_dir)
|
|
return report
|
|
|
|
|
|
def _page_summary(slug: str, wiki_dir: Path) -> tuple[str, str]:
|
|
"""Return (slug, title_from_h1) for index generation."""
|
|
page_path = wiki_dir / f"{slug}.md"
|
|
title = slug
|
|
if page_path.exists():
|
|
first_line = page_path.read_text(encoding="utf-8").splitlines()[0]
|
|
if first_line.startswith("# "):
|
|
title = first_line[2:]
|
|
return (slug, title)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# write_index
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def write_index(
|
|
pages: list[tuple[str, str]],
|
|
*,
|
|
wiki_dir: Path | None = None,
|
|
) -> None:
|
|
"""Generate ``wiki/index.md`` with [[links]] to all compiled concept pages.
|
|
|
|
Parameters
|
|
----------
|
|
pages:
|
|
List of ``(slug, title)`` tuples.
|
|
wiki_dir:
|
|
Path to the wiki directory (defaults to ``config.wiki_dir``).
|
|
"""
|
|
settings = get_settings()
|
|
_wiki_dir = wiki_dir or Path(settings.wiki_dir)
|
|
|
|
ts = datetime.now(UTC).strftime("%Y-%m-%d %H:%M UTC")
|
|
lines = [
|
|
"# Wiki Index",
|
|
"",
|
|
f"_Generated {ts} by `codex wiki compile`_",
|
|
"",
|
|
"## Concepts",
|
|
"",
|
|
]
|
|
for slug, title in sorted(pages, key=lambda x: x[0]):
|
|
lines.append(f"- [[{slug}]] — {title}")
|
|
|
|
lines.append("")
|
|
content = "\n".join(lines)
|
|
(_wiki_dir / "index.md").write_text(content, encoding="utf-8")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# append_log
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def append_log(
|
|
report: CompileReport,
|
|
*,
|
|
wiki_dir: Path | None = None,
|
|
) -> None:
|
|
"""Append a run entry to ``wiki/log.md`` (never overwrites).
|
|
|
|
Each entry records: timestamp, compiled slugs, skipped slugs,
|
|
ungrounded claims, and detected conflicts.
|
|
"""
|
|
settings = get_settings()
|
|
_wiki_dir = wiki_dir or Path(settings.wiki_dir)
|
|
|
|
log_path = _wiki_dir / "log.md"
|
|
ts = report.ran_at.strftime("%Y-%m-%d %H:%M:%S UTC")
|
|
|
|
entry_lines = [
|
|
f"\n## Run {ts}",
|
|
"",
|
|
]
|
|
|
|
if report.compiled:
|
|
entry_lines.append(f"**Compiled:** {', '.join(report.compiled)}")
|
|
if report.skipped:
|
|
entry_lines.append(f"**Skipped (unchanged):** {', '.join(report.skipped)}")
|
|
if report.ungrounded:
|
|
entry_lines.append("")
|
|
entry_lines.append("**⚠ Ungrounded claims:**")
|
|
for slug, text in report.ungrounded:
|
|
entry_lines.append(f"- `{slug}`: {text[:120]}")
|
|
if report.conflicts:
|
|
entry_lines.append("")
|
|
entry_lines.append("**⚠ Conflicts detected:**")
|
|
for slug, bk1, bk2 in report.conflicts:
|
|
entry_lines.append(f"- `{slug}`: conflicting claims in {bk1} vs {bk2}")
|
|
entry_lines.append("")
|
|
|
|
entry = "\n".join(entry_lines)
|
|
|
|
# Append-only: open in append mode
|
|
with log_path.open("a", encoding="utf-8") as fh:
|
|
if log_path.stat().st_size == 0:
|
|
fh.write("# Wiki Compile Log\n")
|
|
fh.write(entry)
|