Merge pull request 'fix: audit remediation Wave 3 — M-1, MED/LOW sweep, ingest robustness + D-1 close-out' (#14) from fix/audit-wave-3 into main

This commit is contained in:
2026-06-16 04:53:59 +00:00
25 changed files with 527 additions and 151 deletions

View File

@@ -5,6 +5,12 @@
# Example: postgresql://researcher:change_me@localhost:5432/papers # Example: postgresql://researcher:change_me@localhost:5432/papers
DATABASE_URL=postgresql://researcher:change_me@localhost:5432/papers DATABASE_URL=postgresql://researcher:change_me@localhost:5432/papers
# Optional: privileged connection used by `codex migrate` to apply schema DDL.
# The app DATABASE_URL is often a least-privilege DML role that cannot CREATE/
# ALTER owner-held tables; point this at an owner/superuser connection.
# Falls back to DATABASE_URL when unset.
# MIGRATION_DATABASE_URL=postgresql://postgres:change_me@localhost:5432/papers
# GROBID service base URL (containerised — see infra/docker-compose.yml) # GROBID service base URL (containerised — see infra/docker-compose.yml)
GROBID_URL=http://localhost:8070 GROBID_URL=http://localhost:8070

View File

@@ -19,13 +19,11 @@ $EDITOR .env
# 3. Install Python dependencies (requires uv) # 3. Install Python dependencies (requires uv)
uv sync uv sync
# 4. Apply the database schema (first run only) # 4. Apply the database schema (idempotent — safe to re-run after upgrades)
uv run python -c " # Needs a role that can run DDL. If DATABASE_URL is a least-privilege app
from codex.db import get_conn, apply_schema # role, point MIGRATION_DATABASE_URL at an owner/superuser connection first:
with get_conn() as conn: # MIGRATION_DATABASE_URL=postgresql://postgres:...@host:5432/papers
apply_schema(conn) uv run codex migrate
print('Schema applied.')
"
``` ```
--- ---
@@ -34,7 +32,8 @@ print('Schema applied.')
| Variable | Default | Description | | Variable | Default | Description |
|---|---|---| |---|---|---|
| `DATABASE_URL` | `postgresql://researcher:change_me@localhost:5432/papers` | libpq connection string | | `DATABASE_URL` | `postgresql://researcher:change_me@localhost:5432/papers` | libpq connection string (app role; DML) |
| `MIGRATION_DATABASE_URL` | *(falls back to `DATABASE_URL`)* | Privileged connection used by `codex migrate` for schema DDL (owner/superuser) |
| `GROBID_URL` | `http://localhost:8070` | GROBID HTTP API base URL | | `GROBID_URL` | `http://localhost:8070` | GROBID HTTP API base URL |
| `OLLAMA_BASE_URL` | `http://localhost:11434` | Local Ollama endpoint (optional) | | `OLLAMA_BASE_URL` | `http://localhost:11434` | Local Ollama endpoint (optional) |
| `EMBEDDING_MODEL` | `BAAI/bge-m3` | sentence-transformers model name | | `EMBEDDING_MODEL` | `BAAI/bge-m3` | sentence-transformers model name |

View File

@@ -29,6 +29,37 @@ app.add_typer(graph_app, name="graph")
app.add_typer(search_app, name="search") app.add_typer(search_app, name="search")
@app.command()
def migrate() -> None:
"""Apply infra/schema.sql to bring the database schema up to date (idempotent).
Must connect as a role that can run DDL (owns / can CREATE + ALTER the tables).
The application's DATABASE_URL is usually a least-privilege DML role and will
fail with InsufficientPrivilege; set MIGRATION_DATABASE_URL to a privileged
(owner/superuser) connection (audit M-1).
"""
import psycopg
import psycopg.rows
from codex.config import get_settings
from codex.db import apply_schema
settings = get_settings()
url = (settings.migration_database_url or settings.database_url).get_secret_value()
try:
with psycopg.connect(url, row_factory=psycopg.rows.dict_row) as conn:
apply_schema(conn)
except psycopg.errors.InsufficientPrivilege as exc:
typer.echo(
"ERROR: schema migration needs a role that owns the tables "
"(CREATE/ALTER). Set MIGRATION_DATABASE_URL to a privileged connection "
f"and retry. ({exc})",
err=True,
)
raise typer.Exit(1) from exc
typer.echo("Schema applied — database is up to date.")
@app.command() @app.command()
def ingest( def ingest(
paper_id: str = typer.Argument(..., help="arXiv ID, DOI, or OpenAlex W-ID"), paper_id: str = typer.Argument(..., help="arXiv ID, DOI, or OpenAlex W-ID"),
@@ -71,7 +102,8 @@ def search_paper(
cite_boost: bool = typer.Option( cite_boost: bool = typer.Option(
False, False,
"--cite-boost", "--cite-boost",
help="Weight results by citation PageRank (F-15). Graceful when corpus < 5 papers.", help="Re-rank the top results by citation PageRank — a within-page "
"tie-breaker, not a hard re-ranking (F-15). Graceful when corpus < 5 papers.",
), ),
) -> None: ) -> None:
"""Semantic similarity search over paper abstracts.""" """Semantic similarity search over paper abstracts."""
@@ -157,6 +189,24 @@ def discover_leads(
typer.echo(f"{item['pull']:>4}× {item['cited_id']}") typer.echo(f"{item['pull']:>4}× {item['cited_id']}")
@discover_app.command("recommend")
def discover_recommend(
paper_id: str = typer.Argument(
..., help="Semantic Scholar paper ID (or arXiv:/DOI: prefixed)."
),
limit: int = typer.Option(20, "--limit", "-n", help="Number of recommendations."),
) -> None:
"""Recommended papers for PAPER_ID via Semantic Scholar."""
from codex.sources.semanticscholar import fetch_recommendations
rec = fetch_recommendations(paper_id, limit=limit)
if not rec:
typer.echo(f"No recommendations found for {paper_id}.")
return
for pid in rec:
typer.echo(pid)
@discover_app.command("citing") @discover_app.command("citing")
def discover_citing(paper_id: str = typer.Argument(...)) -> None: def discover_citing(paper_id: str = typer.Argument(...)) -> None:
"""List papers that cite PAPER_ID.""" """List papers that cite PAPER_ID."""
@@ -391,7 +441,15 @@ def synthesis_leads(
"--lib-path", "--lib-path",
help=( help=(
"Path to a C++ source tree. Enables improvement leads (@cite-aware) and " "Path to a C++ source tree. Enables improvement leads (@cite-aware) and "
"code-vs-corpus gap detection. Without it only connection + topic gaps run." "code-vs-corpus gap detection."
),
),
topic: list[str] | None = typer.Option( # noqa: B008
None,
"--topic",
help=(
"Topic to check for coverage gaps (repeatable). Opt-in: without --topic "
"no coverage-map gaps run (avoids per-paper false positives, audit R-8)."
), ),
), ),
output_dir: Optional[str] = typer.Option( # noqa: UP045 output_dir: Optional[str] = typer.Option( # noqa: UP045
@@ -413,7 +471,7 @@ def synthesis_leads(
llm = default_llm_client() llm = default_llm_client()
connections = find_connections(top_k=settings.synthesis_top_k, llm=llm) connections = find_connections(top_k=settings.synthesis_top_k, llm=llm)
gaps = find_gaps(lib_path=lib_path, llm=llm) gaps = find_gaps(lib_path=lib_path, llm=llm, topics=topic or None)
improvements = ( improvements = (
find_improvements(lib_path, top_k=settings.synthesis_top_k, llm=llm) if lib_path else [] find_improvements(lib_path, top_k=settings.synthesis_top_k, llm=llm) if lib_path else []
) )
@@ -543,7 +601,9 @@ def graph_report(
settings = get_settings() settings = get_settings()
with get_conn() as conn: with get_conn() as conn:
graph = build_citation_graph(conn) graph = build_citation_graph(conn)
known_ids = {row["id"] for row in conn.execute("SELECT id FROM papers").fetchall()} paper_rows = conn.execute("SELECT id, title FROM papers").fetchall()
known_ids = {row["id"] for row in paper_rows}
titles = {row["id"]: row["title"] for row in paper_rows}
if graph.number_of_nodes() == 0: if graph.number_of_nodes() == 0:
typer.echo("Citation graph is empty — ingest some papers first.") typer.echo("Citation graph is empty — ingest some papers first.")
@@ -553,6 +613,13 @@ def graph_report(
pr = citation_pagerank(graph) pr = citation_pagerank(graph)
hubs = sorted(pr.items(), key=lambda x: -x[1])[:top_n] hubs = sorted(pr.items(), key=lambda x: -x[1])[:top_n]
dangling = sorted(dangling_citations(graph, known_ids)) dangling = sorted(dangling_citations(graph, known_ids))
small_corpus = n_papers < settings.graph_min_corpus_size
warning = (
f"only {n_papers} ingested papers; recommended >= {settings.graph_min_corpus_size} "
"for meaningful ranking"
if small_corpus
else None
)
if output_json: if output_json:
typer.echo( typer.echo(
@@ -560,7 +627,11 @@ def graph_report(
{ {
"nodes": graph.number_of_nodes(), "nodes": graph.number_of_nodes(),
"edges": graph.number_of_edges(), "edges": graph.number_of_edges(),
"hubs": [{"paper_id": pid, "pagerank": score} for pid, score in hubs], "small_corpus_warning": warning, # null unless below threshold (audit U-3)
"hubs": [
{"paper_id": pid, "title": titles.get(pid), "pagerank": score}
for pid, score in hubs
],
"dangling_count": len(dangling), "dangling_count": len(dangling),
"dangling": dangling, "dangling": dangling,
}, },
@@ -569,18 +640,16 @@ def graph_report(
) )
return return
if n_papers < settings.graph_min_corpus_size: if warning:
typer.echo( typer.echo(f"Warning: {warning}.", err=True)
f"Warning: only {n_papers} ingested papers "
f"(recommended ≥ {settings.graph_min_corpus_size} for meaningful ranking).",
err=True,
)
typer.echo(f"Graph: {graph.number_of_nodes()} nodes, {graph.number_of_edges()} edges") typer.echo(f"Graph: {graph.number_of_nodes()} nodes, {graph.number_of_edges()} edges")
typer.echo("") typer.echo("")
typer.echo(f"TOP-{top_n} HUB PAPERS (PageRank)") typer.echo(f"TOP-{top_n} HUB PAPERS (PageRank)")
typer.echo("-" * 48) typer.echo("-" * 48)
for pid, score in hubs: for pid, score in hubs:
typer.echo(f" {score:.4f} {pid}") # in-KB hubs show their title; dangling (OpenAlex-id) hubs are flagged (U-1)
label = titles.get(pid) or "(dangling — not ingested)"
typer.echo(f" {score:.4f} {pid} {label}")
typer.echo("") typer.echo("")
typer.echo(f"DANGLING CITATIONS ({len(dangling)} cited but not ingested)") typer.echo(f"DANGLING CITATIONS ({len(dangling)} cited but not ingested)")
typer.echo("-" * 48) typer.echo("-" * 48)
@@ -613,3 +682,23 @@ def graph_related(
) )
for pid in related: for pid in related:
typer.echo(f" {pid}") typer.echo(f" {pid}")
@graph_app.command("cocited")
def graph_cocited(
paper_id: str = typer.Argument(..., help="Paper ID to find co-cited papers for."),
) -> None:
"""List papers frequently co-cited with PAPER_ID (graph co-citation)."""
from codex.db import get_conn
from codex.graph import build_citation_graph, find_co_cited
with get_conn() as conn:
graph = build_citation_graph(conn)
results = find_co_cited(paper_id, graph)
if not results:
typer.echo(f"No papers co-cited with {paper_id}.")
return
typer.echo(f"Papers co-cited with {paper_id} (count desc):")
for pid, count in results:
typer.echo(f" {count:>3}× {pid}")

View File

@@ -9,7 +9,7 @@ from __future__ import annotations
from functools import lru_cache from functools import lru_cache
from pydantic import AliasChoices, Field from pydantic import AliasChoices, Field, SecretStr
from pydantic_settings import BaseSettings, SettingsConfigDict from pydantic_settings import BaseSettings, SettingsConfigDict
@@ -26,14 +26,25 @@ class Settings(BaseSettings):
# ------------------------------------------------------------------ # ------------------------------------------------------------------
# Database # Database
# ------------------------------------------------------------------ # ------------------------------------------------------------------
database_url: str = Field( database_url: SecretStr = Field(
default="postgresql://researcher:change_me@localhost:5432/papers", default=SecretStr("postgresql://researcher:change_me@localhost:5432/papers"),
description=( description=(
"libpq-compatible connection string consumed by psycopg. " "libpq-compatible connection string consumed by psycopg. "
"Example: postgresql://user:pass@host:5432/dbname" "Example: postgresql://user:pass@host:5432/dbname"
), ),
) )
migration_database_url: SecretStr | None = Field(
default=None,
description=(
"Optional privileged connection string used by `codex migrate` to apply "
"schema DDL. The app's database_url is typically a least-privilege DML "
"role that cannot CREATE/ALTER owner-held tables (raises "
"InsufficientPrivilege); point this at an owner/superuser connection. "
"Falls back to database_url when unset."
),
)
# ------------------------------------------------------------------ # ------------------------------------------------------------------
# External services # External services
# ------------------------------------------------------------------ # ------------------------------------------------------------------
@@ -137,7 +148,7 @@ class Settings(BaseSettings):
# ------------------------------------------------------------------ # ------------------------------------------------------------------
# F-09 Rich Parsing # F-09 Rich Parsing
# ------------------------------------------------------------------ # ------------------------------------------------------------------
mathpix_app_id: str | None = Field( mathpix_app_id: SecretStr | None = Field(
default=None, default=None,
description=( description=(
"MathPix App ID for cloud formula extraction. " "MathPix App ID for cloud formula extraction. "
@@ -145,7 +156,7 @@ class Settings(BaseSettings):
), ),
) )
mathpix_app_key: str | None = Field( mathpix_app_key: SecretStr | None = Field(
default=None, default=None,
description=( description=(
"MathPix App Key for cloud formula extraction. " "MathPix App Key for cloud formula extraction. "
@@ -246,7 +257,7 @@ class Settings(BaseSettings):
description="Bind port for HTTP transport (ignored for stdio).", description="Bind port for HTTP transport (ignored for stdio).",
) )
mcp_auth_token: str | None = Field( mcp_auth_token: SecretStr | None = Field(
default=None, default=None,
description=( description=(
"Bearer token required when mcp_transport='http'. " "Bearer token required when mcp_transport='http'. "

View File

@@ -36,7 +36,7 @@ def get_conn() -> Generator[psycopg.Connection[psycopg.rows.DictRow], None, None
""" """
settings = get_settings() settings = get_settings()
with psycopg.connect( with psycopg.connect(
settings.database_url, settings.database_url.get_secret_value(),
row_factory=psycopg.rows.dict_row, row_factory=psycopg.rows.dict_row,
) as conn: ) as conn:
yield conn yield conn
@@ -45,11 +45,16 @@ def get_conn() -> Generator[psycopg.Connection[psycopg.rows.DictRow], None, None
def apply_schema(conn: psycopg.Connection[psycopg.rows.DictRow]) -> None: def apply_schema(conn: psycopg.Connection[psycopg.rows.DictRow]) -> None:
"""Idempotently apply ``infra/schema.sql`` to the connected database. """Idempotently apply ``infra/schema.sql`` to the connected database.
Safe to call on every startup — all DDL statements use Safe to call repeatedly — every statement is ``CREATE … IF NOT EXISTS`` or
``CREATE … IF NOT EXISTS``. ``ADD COLUMN IF NOT EXISTS``, so re-application is a no-op.
Requires a connection whose role can run DDL (owns / can CREATE + ALTER the
tables). The least-privilege application role is typically DML-only and will
raise ``InsufficientPrivilege``; use a privileged connection — see
``codex migrate`` and ``MIGRATION_DATABASE_URL`` (audit M-1).
Args: Args:
conn: An open psycopg connection (obtained via :func:`get_conn`). conn: An open psycopg connection with DDL privileges.
""" """
schema_path = Path(__file__).parent.parent / "infra" / "schema.sql" schema_path = Path(__file__).parent.parent / "infra" / "schema.sql"
sql = schema_path.read_text(encoding="utf-8") sql = schema_path.read_text(encoding="utf-8")

View File

@@ -1,4 +1,4 @@
"""Hybrid dense + sparse embeddings via BGE-M3 (ADR-0002). """Dense (+ optional sparse) embeddings via BGE-M3 (ADR-0002).
Uses :class:`FlagEmbedding.BGEM3FlagModel` for encoding because the Uses :class:`FlagEmbedding.BGEM3FlagModel` for encoding because the
``return_dense`` / ``return_sparse`` kwargs are part of the FlagEmbedding ``return_dense`` / ``return_sparse`` kwargs are part of the FlagEmbedding
@@ -7,6 +7,11 @@ vanilla ``SentenceTransformer`` load of ``BAAI/bge-m3`` (same weights,
same model); sparse output is a list of ``{token_id: weight}`` dicts per same model); sparse output is a list of ``{token_id: weight}`` dicts per
text. text.
Status: only the **dense** path is wired into ingest/search today; the search
"hybrid" is dense + Postgres FTS. :meth:`Embedder.encode_sparse` / :meth:`encode`
are provided for a future sparse-retrieval layer but are not yet consumed by the
pipeline (audit C-14).
Notes for callers Notes for callers
----------------- -----------------
* Empty input is handled explicitly — no model call is issued. * Empty input is handled explicitly — no model call is issued.

View File

@@ -8,6 +8,7 @@ from dataclasses import dataclass
from pathlib import Path from pathlib import Path
import numpy as np import numpy as np
import psycopg
from codex.config import get_settings from codex.config import get_settings
from codex.db import get_conn from codex.db import get_conn
@@ -125,35 +126,64 @@ def ingest_paper(
# --------------------------------------------------------------- # ---------------------------------------------------------------
# 3. Upsert paper (INSERT … ON CONFLICT (id) DO UPDATE SET …) # 3. Upsert paper (INSERT … ON CONFLICT (id) DO UPDATE SET …)
# --------------------------------------------------------------- # ---------------------------------------------------------------
upsert_sql = """
INSERT INTO papers (id, openalex_id, bibkey, title, authors, year, abstract,
source_path, abstract_emb)
VALUES (%(id)s, %(openalex_id)s, %(bibkey)s, %(title)s, %(authors)s, %(year)s,
%(abstract)s, %(source_path)s, %(abstract_emb)s)
ON CONFLICT (id) DO UPDATE SET
openalex_id = EXCLUDED.openalex_id,
bibkey = COALESCE(papers.bibkey, EXCLUDED.bibkey),
title = EXCLUDED.title,
authors = EXCLUDED.authors,
year = EXCLUDED.year,
abstract = EXCLUDED.abstract,
source_path = COALESCE(EXCLUDED.source_path, papers.source_path),
abstract_emb = EXCLUDED.abstract_emb
"""
with get_conn() as conn: with get_conn() as conn:
conn.execute( # The id-keyed upsert does not catch a papers.bibkey UNIQUE collision (two
""" # same-author/year papers generate the same key). On such a collision retry
INSERT INTO papers (id, openalex_id, bibkey, title, authors, year, abstract, # with an a/b/c… suffix instead of failing the whole ingest (audit C-15).
source_path, abstract_emb) base_bibkey = paper.bibkey
VALUES (%(id)s, %(openalex_id)s, %(bibkey)s, %(title)s, %(authors)s, %(year)s, for attempt in range(27):
%(abstract)s, %(source_path)s, %(abstract_emb)s) try:
ON CONFLICT (id) DO UPDATE SET conn.execute(
openalex_id = EXCLUDED.openalex_id, upsert_sql,
bibkey = COALESCE(papers.bibkey, EXCLUDED.bibkey), {
title = EXCLUDED.title, "id": paper.id,
authors = EXCLUDED.authors, "openalex_id": paper.openalex_id,
year = EXCLUDED.year, "bibkey": paper.bibkey,
abstract = EXCLUDED.abstract, "title": paper.title,
source_path = COALESCE(EXCLUDED.source_path, papers.source_path), "authors": paper.authors,
abstract_emb = EXCLUDED.abstract_emb "year": paper.year,
""", "abstract": paper.abstract,
{ "source_path": source_path,
"id": paper.id, "abstract_emb": abstract_emb,
"openalex_id": paper.openalex_id, },
"bibkey": paper.bibkey, )
"title": paper.title, break
"authors": paper.authors, except psycopg.errors.UniqueViolation as exc:
"year": paper.year, msg = str(exc).lower()
"abstract": paper.abstract, conn.rollback()
"source_path": source_path, if "openalex" in msg:
"abstract_emb": abstract_emb, # Two different papers.id claim the same OpenAlex work — a real data
}, # conflict, not something to auto-rename. Surface it clearly instead
) # of the raw psycopg error (companion gap to C-15's bibkey case).
raise ValueError(
f"openalex_id {paper.openalex_id!r} is already attached to a "
f"different paper; ingesting '{paper.id}' would duplicate it — "
"resolve the conflicting paper first."
) from exc
if base_bibkey is None or "bibkey" not in msg or attempt == 26:
raise
paper.bibkey = f"{base_bibkey}{chr(ord('a') + attempt)}"
logger.warning(
"bibkey %r already exists; ingesting '%s' as %r instead",
base_bibkey,
paper.id,
paper.bibkey,
)
# Register openalex_id in paper_identifiers (alias table). # Register openalex_id in paper_identifiers (alias table).
# Every ingest records the primary openalex_id so the graph JOIN # Every ingest records the primary openalex_id so the graph JOIN

View File

@@ -24,6 +24,7 @@ Transport
from __future__ import annotations from __future__ import annotations
import logging import logging
import secrets
from mcp.server.fastmcp import FastMCP from mcp.server.fastmcp import FastMCP
from mcp.server.transport_security import TransportSecuritySettings from mcp.server.transport_security import TransportSecuritySettings
@@ -121,7 +122,6 @@ def wiki_read(concept_slug: str) -> dict[str, object]:
directory or the requested page does not exist. directory or the requested page does not exist.
""" """
try: try:
import json
from pathlib import Path from pathlib import Path
from codex.config import get_settings from codex.config import get_settings
@@ -135,16 +135,12 @@ def wiki_read(concept_slug: str) -> dict[str, object]:
markdown = page_path.read_text(encoding="utf-8") markdown = page_path.read_text(encoding="utf-8")
# Collect cited bibkeys from the compile state # Collect the actual cited bibkeys from the page's inline [BibKey #loc]
state_path = wiki_dir / ".compile-state.json" # citations (audit C-3: this previously returned the concept slug itself,
sources: list[str] = [] # not the cited sources).
if state_path.exists(): from codex.wiki import _parse_claims
try:
state: dict[str, str] = json.loads(state_path.read_text(encoding="utf-8")) sources = sorted({claim.bibkey for claim in _parse_claims(markdown)})
if concept_slug in state:
sources = [concept_slug]
except (json.JSONDecodeError, OSError):
pass
return {"markdown": markdown, "sources": sources} return {"markdown": markdown, "sources": sources}
except Exception as exc: # noqa: BLE001 except Exception as exc: # noqa: BLE001
@@ -298,12 +294,13 @@ def _require_http_token() -> str:
from codex.config import get_settings from codex.config import get_settings
settings = get_settings() settings = get_settings()
if not settings.mcp_auth_token: token = settings.mcp_auth_token.get_secret_value() if settings.mcp_auth_token else ""
if not token:
raise RuntimeError( raise RuntimeError(
"HTTP transport requires MCP_AUTH_TOKEN to be set. " "HTTP transport requires MCP_AUTH_TOKEN to be set. "
"Set the environment variable or add it to .env." "Set the environment variable or add it to .env."
) )
return settings.mcp_auth_token return token
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -337,7 +334,9 @@ def main() -> None:
self, request: Request, call_next: RequestResponseEndpoint self, request: Request, call_next: RequestResponseEndpoint
) -> Response: ) -> Response:
auth = request.headers.get("Authorization", "") auth = request.headers.get("Authorization", "")
if auth != f"Bearer {token}": # Constant-time comparison to avoid a token-length/equality timing
# side-channel (audit S-2).
if not secrets.compare_digest(auth, f"Bearer {token}"):
return Response("Unauthorized", status_code=401) return Response("Unauthorized", status_code=401)
return await call_next(request) return await call_next(request)

View File

@@ -302,8 +302,10 @@ def extract_formulas(
""" """
settings = get_settings() settings = get_settings()
app_id = mathpix_app_id or settings.mathpix_app_id or "" settings_id = settings.mathpix_app_id.get_secret_value() if settings.mathpix_app_id else ""
app_key = mathpix_app_key or settings.mathpix_app_key or "" settings_key = settings.mathpix_app_key.get_secret_value() if settings.mathpix_app_key else ""
app_id = mathpix_app_id or settings_id
app_key = mathpix_app_key or settings_key
if app_id and app_key: if app_id and app_key:
logger.info("Using MathPix backend for %s", pdf_path) logger.info("Using MathPix backend for %s", pdf_path)

View File

@@ -38,10 +38,7 @@ def fetch_source(arxiv_id: str) -> str | None:
file is present (signals Nougat fallback). file is present (signals Nougat fallback).
""" """
url = f"{_BASE}/src/{arxiv_id}" url = f"{_BASE}/src/{arxiv_id}"
try: response = httpx.get(url, timeout=60, follow_redirects=True)
response = httpx.get(url, timeout=60, follow_redirects=True)
except httpx.RequestError:
raise
if response.status_code == 404: if response.status_code == 404:
logger.debug("arXiv 404 for source id=%s", arxiv_id) logger.debug("arXiv 404 for source id=%s", arxiv_id)
return None return None

View File

@@ -14,6 +14,7 @@ import logging
import threading import threading
import time import time
from typing import Any from typing import Any
from urllib.parse import quote
import httpx import httpx
from tenacity import retry, retry_if_exception, stop_after_attempt, wait_exponential from tenacity import retry, retry_if_exception, stop_after_attempt, wait_exponential
@@ -77,7 +78,9 @@ def fetch_references(paper_id: str) -> list[Citation]:
list[Citation] list[Citation]
One Citation per reference, with optional context snippet. One Citation per reference, with optional context snippet.
""" """
url = f"{_BASE_GRAPH}/paper/{paper_id}/references" # quote the id so a legacy-arXiv "/" (e.g. arXiv:math/0603097) doesn't break
# the URL path; keep ":" for the arXiv:/DOI: scheme prefix (audit R-5).
url = f"{_BASE_GRAPH}/paper/{quote(paper_id, safe=':')}/references"
params: dict[str, Any] = {"fields": "externalIds,contexts"} params: dict[str, Any] = {"fields": "externalIds,contexts"}
try: try:
response = _get(url, params=params) response = _get(url, params=params)
@@ -118,7 +121,7 @@ def fetch_recommendations(paper_id: str, limit: int = 20) -> list[str]:
list[str] list[str]
List of recommended paper IDs. List of recommended paper IDs.
""" """
url = f"{_BASE_RECS}/papers/forpaper/{paper_id}" url = f"{_BASE_RECS}/papers/forpaper/{quote(paper_id, safe=':')}"
params: dict[str, Any] = {"limit": limit} params: dict[str, Any] = {"limit": limit}
try: try:
response = _get(url, params=params) response = _get(url, params=params)

View File

@@ -416,11 +416,12 @@ def find_gaps(
Two complementary signals are used: Two complementary signals are used:
* **Topic coverage map** — for each topic in ``topics`` (or, if * **Topic coverage map** — for each **explicitly-requested** topic in
``None``, the union of paper titles), retrieve chunks and check ``topics``, retrieve chunks and check bibkey coverage. A topic covered by
bibkey coverage. A topic covered by < ``synthesis_gap_min_coverage`` < ``synthesis_gap_min_coverage`` bibkeys is a gap candidate and the LLM is
bibkeys is reported as a gap candidate and the LLM is asked to asked to articulate it. Defaulting to paper titles flagged almost every
articulate it. paper as a gap (a title retrieves mostly its own single bibkey), so the
coverage map is opt-in now; pass ``topics`` / CLI ``--topic`` (audit R-8).
* **Code-vs-corpus gap** — when ``lib_path`` is given, any @cite * **Code-vs-corpus gap** — when ``lib_path`` is given, any @cite
bibkey scanned from code that is NOT present in the corpus is bibkey scanned from code that is NOT present in the corpus is
flagged. This catches the "cited in code, never ingested" case. flagged. This catches the "cited in code, never ingested" case.
@@ -429,9 +430,9 @@ def find_gaps(
seq = 1 seq = 1
leads: list[Lead] = [] leads: list[Lead] = []
# Coverage-map probes are the explicit topics only — no default-to-titles,
# which turned every paper into a spurious gap (audit R-8).
probe_topics: list[str] = list(topics or []) probe_topics: list[str] = list(topics or [])
if not probe_topics:
probe_topics = [title for _, title in _list_paper_titles(db_conn)]
known_bibkeys = {bk for bk, _ in _list_paper_titles(db_conn)} known_bibkeys = {bk for bk, _ in _list_paper_titles(db_conn)}

View File

@@ -186,8 +186,15 @@ def _retrieve_chunks(
).fetchall() ).fetchall()
# Filter reference-list chunks: skip chunks whose content looks like a bibliography # Filter reference-list chunks: skip chunks whose content looks like a bibliography
# (heuristic: > 60 % of lines match "^\[\d+\]" or "^[A-Z][a-z]+,?\s+[A-Z]\."). # (heuristic: > 60 % of lines look like "[12]" refs or "Surname, I." authors).
ref_pattern = re.compile(r"^\s*(\[\d+\]|[A-Z][a-z]+,?\s+[A-Z]\.)", re.MULTILINE) # The author alternative excludes common math-structure words ("Theorem A.",
# "Lemma B.") so theorem-dense chunks aren't mistaken for a reference list (R-2).
ref_pattern = re.compile(
r"^\s*(\[\d+\]|(?!(?:Theorem|Lemma|Proposition|Corollary|Definition|Proof|"
r"Section|Figure|Fig|Table|Equation|Eq|Remark|Example|Chapter)\b)"
r"[A-Z][a-z]+,?\s+[A-Z]\.)",
re.MULTILINE,
)
filtered: list[dict[str, Any]] = [] filtered: list[dict[str, Any]] = []
for row in rows: for row in rows:
content: str = row["content"] content: str = row["content"]
@@ -273,7 +280,13 @@ _STOPWORDS: frozenset[str] = frozenset(
} }
) )
_SENTENCE_SPLIT_RE = re.compile(r"[.!?]") # Split on sentence-ending punctuation only when followed by whitespace/end, so a
# decimal like "1.5" does not create a spurious sentence boundary (audit R-1).
_SENTENCE_SPLIT_RE = re.compile(r"[.!?]+(?=\s|$)")
# Punctuation stripped from token edges before grounding comparison — sentence/
# clause marks and quotes, NOT math delimiters like ()=+ (audit R-1).
_EDGE_PUNCT = ".,;:!?\"'"
def _last_sentence(text: str) -> str: def _last_sentence(text: str) -> str:
@@ -292,8 +305,16 @@ def _last_sentence(text: str) -> str:
def _content_words(text: str) -> list[str]: def _content_words(text: str) -> list[str]:
"""Return lowercased tokens with stopwords removed.""" """Return lowercased content tokens, edge-punctuation-stripped, stopwords removed.
return [w for w in text.lower().split() if w not in _STOPWORDS]
Stripping leading/trailing sentence punctuation makes grounding robust to
surface variation ("volume," vs "volume") so a faithful paraphrase is not
marked ungrounded over a comma (audit R-1). Math delimiters (``()=+``) are
preserved. Claim and chunk tokens are normalised identically, so matching
stays consistent.
"""
tokens = (w.strip(_EDGE_PUNCT) for w in text.lower().split())
return [w for w in tokens if w and w not in _STOPWORDS]
def _run_grounding_guard( def _run_grounding_guard(
@@ -322,8 +343,9 @@ def _run_grounding_guard(
for chunk in chunks: for chunk in chunks:
bk = str(chunk.get("bibkey") or "") bk = str(chunk.get("bibkey") or "")
if bk: if bk:
# Content words of the chunk (stopwords removed, lowercased) # Content words of the chunk, space-padded so a 5-gram match is bounded
cw = " ".join(_content_words(chunk["content"])) # by whole tokens, not a substring bleeding across words (audit C-6).
cw = " " + " ".join(_content_words(chunk["content"])) + " "
bib_index.setdefault(bk, []).append(cw) bib_index.setdefault(bk, []).append(cw)
for claim in claims: for claim in claims:
@@ -345,7 +367,7 @@ def _run_grounding_guard(
found = False found = False
for i in range(len(content) - 4): # content-5-grams for i in range(len(content) - 4): # content-5-grams
gram = " ".join(content[i : i + 5]) gram = " " + " ".join(content[i : i + 5]) + " " # padded: token-boundary match
if any(gram in src for src in sources): if any(gram in src for src in sources):
found = True found = True
break break
@@ -402,6 +424,13 @@ def _detect_conflicts(
_INLINE_CODE_RE = re.compile(r"`[^`]+`") _INLINE_CODE_RE = re.compile(r"`[^`]+`")
# LaTeX math spans — protected from cross-ref injection so a concept name inside
# a formula is not rewritten to a [[slug]] link, corrupting the LaTeX (audit R-9).
_MATH_SPAN_RE = re.compile(
r"\$\$.*?\$\$|\$[^$\n]*?\$|\\\(.*?\\\)|\\\[.*?\\\]",
re.DOTALL,
)
def _inject_cross_refs( def _inject_cross_refs(
markdown: str, markdown: str,
@@ -411,11 +440,12 @@ def _inject_cross_refs(
"""Replace occurrences of other concept titles/aliases with ``[[slug]]`` links. """Replace occurrences of other concept titles/aliases with ``[[slug]]`` links.
Only exact case-insensitive whole-word matches outside of existing Only exact case-insensitive whole-word matches outside of existing
``[[…]]`` blocks or inline code spans are replaced. ``[[…]]`` blocks, inline code spans, or LaTeX math spans are replaced.
Inline-code spans (`` `…` ``) are temporarily protected by null-byte Inline-code (`` `…` ``) and math (``$…$``, ``$$…$$``, ``\\(…\\)``,
placeholders and restored after injection. ``\\[…\\]``) spans are temporarily protected by null-byte placeholders and
restored after injection.
""" """
# Step 1: protect inline-code spans from replacement # Step 1: protect inline-code and math spans from replacement
placeholders: dict[str, str] = {} placeholders: dict[str, str] = {}
def _protect(m: re.Match[str]) -> str: def _protect(m: re.Match[str]) -> str:
@@ -424,6 +454,7 @@ def _inject_cross_refs(
return key return key
markdown = _INLINE_CODE_RE.sub(_protect, markdown) markdown = _INLINE_CODE_RE.sub(_protect, markdown)
markdown = _MATH_SPAN_RE.sub(_protect, markdown) # don't rewrite names inside LaTeX (R-9)
# Step 2: inject cross-refs on unprotected text # Step 2: inject cross-refs on unprotected text
for concept in all_concepts: for concept in all_concepts:
@@ -602,14 +633,14 @@ def compile_concept(
claims = _parse_claims(raw_output) claims = _parse_claims(raw_output)
claims = _run_grounding_guard(claims, chunks) claims = _run_grounding_guard(claims, chunks)
_all_concepts = all_concepts or [] # Graceful: formula-embedding hook (F-09) — currently a no-op (audit C-12).
raw_output = _inject_cross_refs(raw_output, _all_concepts, concept.slug)
# Graceful: try to embed formula chunks (F-09) — skip if table missing
_try_embed_formulas(concept, chunks) _try_embed_formulas(concept, chunks)
compiled_at = datetime.now(UTC) compiled_at = datetime.now(UTC)
# Mark ungrounded claims on the raw body FIRST, then inject cross-refs: a
# concept name inside a claim must not stop the ⚠ regex from matching (C-13).
markdown = _render_page_markdown(concept, raw_output, claims, compiled_at) markdown = _render_page_markdown(concept, raw_output, claims, compiled_at)
markdown = _inject_cross_refs(markdown, all_concepts or [], concept.slug)
return ConceptPage( return ConceptPage(
concept=concept, concept=concept,
@@ -621,20 +652,13 @@ def compile_concept(
def _try_embed_formulas(concept: Concept, chunks: list[dict[str, Any]]) -> None: def _try_embed_formulas(concept: Concept, chunks: list[dict[str, Any]]) -> None:
"""Attempt to look up formula chunks for the concept — graceful no-op if F-09 absent.""" """Reserved hook for embedding F-09 formula chunks into a page.
try:
from codex.db import get_conn
with get_conn() as conn: Currently a no-op — formula embedding is not implemented yet. Kept as a stable
# Check if formulas table exists seam for compile_concept's call site. The previous body issued a per-concept
row = conn.execute( DB round-trip (information_schema lookup) that did nothing (audit C-12).
"SELECT 1 FROM information_schema.tables WHERE table_name = 'formulas'" """
).fetchone() return
if row is None:
return # F-09 not present
# (Future: embed relevant raw_latex into the page)
except Exception: # noqa: BLE001
return # DB not reachable or other error — degrade gracefully
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------

View File

@@ -177,13 +177,28 @@ was measured on a graph the current code no longer produces:
diverge (audit C-1). Ingest stores the caller's canonical id as `papers.id` diverge (audit C-1). Ingest stores the caller's canonical id as `papers.id`
instead of OpenAlex's URL-form doi (audit C-7). Inter-KB citations are instead of OpenAlex's URL-form doi (audit C-7). Inter-KB citations are
therefore represented as edges — the "Known Limitation" above is obsolete. therefore represented as edges — the "Known Limitation" above is obsolete.
- **Spike numbers are stale.** The 478 dangling nodes, the rank-7 - **Spike re-measured after migration (2026-06-15).** The corpus was re-ingested
Bobenko/Springborn result, and the 5.7 % score spread were measured *before* onto canonical bare ids (29 papers, all bare — verified `url_form_ids = 0`), and
the resolution + canonicalisation. They must be **re-measured after the corpus the graph re-run on the post-migration corpus:
is re-ingested** onto canonical ids (the chosen C-7 migration); until then the
GO rests on a superseded topology.
**Action (pending):** after re-ingesting via `ingest_all.sh`, re-run | Metric | Original spike (pre-fix) | Post-migration (re-measured) |
`codex graph report`, refresh the Spike Result table, and confirm the |--------|--------------------------|------------------------------|
foundational hubs (Pinkall/Polthier, Yamabe, Schoen, Bobenko/Springborn) still | Ingested papers | 29 | 29 |
surface. | Graph nodes | 495 | 489 |
| Graph edges | 590 | 590 |
| Dangling nodes | 478 | 472 |
| Top-10 PR spread | 5.7 % | 5.4 % |
| Rank-1 hub | W1974956622 (dangling) | **Pinkall/Polthier 1993 — in-KB (resolved)** |
| Bobenko/Springborn 2007 | rank 7 | rank 7 |
The resolution fix's signature is visible: Pinkall/Polthier 1993 now resolves to
its canonical DOI and rises to **rank 1 as an in-KB hub** (previously a dangling
OpenAlex node). Bobenko/Springborn 2007 holds rank 7, and the foundational works
(Pinkall/Polthier, Yamabe, Schoen, Bobenko/Springborn) still surface. **GO
re-affirmed** on the shipping topology.
Migration footnotes: two papers initially survived in URL form because the
id-keyed upsert did not catch a `papers.openalex_id` collision, and `2305.10988`
failed on a `papers.bibkey` collision with `2310.17529` (both →
`BobenkoLutz2023`). Same upsert gap — resolved by the bibkey-retry (audit C-15)
plus a delete + re-ingest of the two stragglers; corpus is now 29/29 canonical.

View File

@@ -76,11 +76,13 @@ PYEOF
)" )"
if [[ "$SECTION_OK" != "yes" ]]; then if [[ "$SECTION_OK" != "yes" ]]; then
echo "ABORT — schema not ready: chunks.section is missing (audit M-1)." echo "ABORT — schema not ready: chunks.section is missing (audit M-1)."
echo "The app user cannot add it (chunks is owned by 'postgres'). Apply it as" echo "The app user cannot add it (core tables are owned by 'postgres'). Sync the"
echo "the table owner, then re-run this script:" echo "schema with a privileged role, then re-run this script. Either:"
echo "" echo ""
echo " ssh alfred@192.168.178.103 \\" echo " (a) MIGRATION_DATABASE_URL=postgresql://postgres:PASS@HOST:PORT/papers \\"
echo " \"sudo -u postgres psql -d papers -c \\\\\"ALTER TABLE chunks ADD COLUMN IF NOT EXISTS section TEXT;\\\\\"\"" echo " \"$CODEX_DIR/.venv/bin/codex\" migrate # applies schema.sql idempotently"
echo " (b) sudo -u postgres psql -d papers \\"
echo " -c \"ALTER TABLE chunks ADD COLUMN IF NOT EXISTS section TEXT;\""
echo "" echo ""
echo "Nothing was changed — no TRUNCATE was issued." echo "Nothing was changed — no TRUNCATE was issued."
exit 1 exit 1

View File

@@ -13,7 +13,7 @@ CREATE EXTENSION IF NOT EXISTS vector;
-- --------------------------------------------------------------------- -- ---------------------------------------------------------------------
-- Layer 1: Papers + full-text chunks -- Layer 1: Papers + full-text chunks
-- --------------------------------------------------------------------- -- ---------------------------------------------------------------------
CREATE TABLE papers ( CREATE TABLE IF NOT EXISTS papers (
id TEXT PRIMARY KEY, -- canonical ID: arXiv ID or DOI id TEXT PRIMARY KEY, -- canonical ID: arXiv ID or DOI
openalex_id TEXT UNIQUE, -- e.g. W2741809807 openalex_id TEXT UNIQUE, -- e.g. W2741809807
bibkey TEXT UNIQUE, -- BibTeX key -> .bib + Doxygen @cite bibkey TEXT UNIQUE, -- BibTeX key -> .bib + Doxygen @cite
@@ -27,10 +27,10 @@ CREATE TABLE papers (
); );
-- HNSW index for fast approximate nearest-neighbour search on abstracts. -- HNSW index for fast approximate nearest-neighbour search on abstracts.
CREATE INDEX papers_abstract_emb_idx CREATE INDEX IF NOT EXISTS papers_abstract_emb_idx
ON papers USING hnsw (abstract_emb vector_cosine_ops); ON papers USING hnsw (abstract_emb vector_cosine_ops);
CREATE TABLE chunks ( CREATE TABLE IF NOT EXISTS chunks (
id BIGSERIAL PRIMARY KEY, id BIGSERIAL PRIMARY KEY,
paper_id TEXT REFERENCES papers(id) ON DELETE CASCADE, paper_id TEXT REFERENCES papers(id) ON DELETE CASCADE,
ord INT NOT NULL, -- position within the paper ord INT NOT NULL, -- position within the paper
@@ -38,13 +38,13 @@ CREATE TABLE chunks (
embedding vector(1024) embedding vector(1024)
); );
CREATE INDEX chunks_emb_idx CREATE INDEX IF NOT EXISTS chunks_emb_idx
ON chunks USING hnsw (embedding vector_cosine_ops); ON chunks USING hnsw (embedding vector_cosine_ops);
CREATE INDEX chunks_paper_idx ON chunks (paper_id); CREATE INDEX IF NOT EXISTS chunks_paper_idx ON chunks (paper_id);
-- Sparse / keyword hits for exact mathematical terminology (hybrid search). -- Sparse / keyword hits for exact mathematical terminology (hybrid search).
-- Dense (above) + full-text (below) combined = robust against math terms. -- Dense (above) + full-text (below) combined = robust against math terms.
CREATE INDEX chunks_fts_idx CREATE INDEX IF NOT EXISTS chunks_fts_idx
ON chunks USING gin (to_tsvector('english', content)); ON chunks USING gin (to_tsvector('english', content));
-- --------------------------------------------------------------------- -- ---------------------------------------------------------------------
@@ -53,19 +53,19 @@ CREATE INDEX chunks_fts_idx
-- edges to not-yet-ingested papers are intentionally preserved — -- edges to not-yet-ingested papers are intentionally preserved —
-- those are your discovery leads. -- those are your discovery leads.
-- --------------------------------------------------------------------- -- ---------------------------------------------------------------------
CREATE TABLE citations ( CREATE TABLE IF NOT EXISTS citations (
citing_id TEXT REFERENCES papers(id) ON DELETE CASCADE, citing_id TEXT REFERENCES papers(id) ON DELETE CASCADE,
cited_id TEXT NOT NULL, -- arXiv/DOI/OpenAlex ID of the target cited_id TEXT NOT NULL, -- arXiv/DOI/OpenAlex ID of the target
context TEXT, -- optional: citation context (S2) context TEXT, -- optional: citation context (S2)
PRIMARY KEY (citing_id, cited_id) PRIMARY KEY (citing_id, cited_id)
); );
CREATE INDEX citations_cited_idx ON citations (cited_id); CREATE INDEX IF NOT EXISTS citations_cited_idx ON citations (cited_id);
-- --------------------------------------------------------------------- -- ---------------------------------------------------------------------
-- Layer 3: Provenance (the "cleanly couple" goal) -- Layer 3: Provenance (the "cleanly couple" goal)
-- --------------------------------------------------------------------- -- ---------------------------------------------------------------------
CREATE TABLE code_links ( CREATE TABLE IF NOT EXISTS code_links (
id BIGSERIAL PRIMARY KEY, id BIGSERIAL PRIMARY KEY,
symbol TEXT NOT NULL, -- e.g. 'dec::hodge_star' or 'file.cpp:120' symbol TEXT NOT NULL, -- e.g. 'dec::hodge_star' or 'file.cpp:120'
paper_id TEXT REFERENCES papers(id) ON DELETE SET NULL, paper_id TEXT REFERENCES papers(id) ON DELETE SET NULL,
@@ -74,8 +74,8 @@ CREATE TABLE code_links (
added_at TIMESTAMPTZ DEFAULT now() added_at TIMESTAMPTZ DEFAULT now()
); );
CREATE INDEX code_links_symbol_idx ON code_links (symbol); CREATE INDEX IF NOT EXISTS code_links_symbol_idx ON code_links (symbol);
CREATE INDEX code_links_paper_idx ON code_links (paper_id); CREATE INDEX IF NOT EXISTS code_links_paper_idx ON code_links (paper_id);
-- --------------------------------------------------------------------- -- ---------------------------------------------------------------------
-- Example query: discovery leads -- Example query: discovery leads

View File

@@ -28,9 +28,11 @@ def _make_conn_cm(conn: MagicMock):
def _make_conn_with_paper_ids(*paper_ids: str) -> MagicMock: def _make_conn_with_paper_ids(*paper_ids: str) -> MagicMock:
"""Mock conn that returns paper rows for SELECT id FROM papers.""" """Mock conn that returns paper rows for SELECT id, title FROM papers."""
conn = MagicMock() conn = MagicMock()
conn.execute.return_value.fetchall.return_value = [{"id": pid} for pid in paper_ids] conn.execute.return_value.fetchall.return_value = [
{"id": pid, "title": f"Title of {pid}"} for pid in paper_ids
]
return conn return conn

View File

@@ -11,6 +11,7 @@ from typing import Any
from unittest.mock import MagicMock, patch from unittest.mock import MagicMock, patch
import numpy as np import numpy as np
import psycopg
import pytest import pytest
from codex.ingest import IngestResult, _make_bibkey, ingest_paper from codex.ingest import IngestResult, _make_bibkey, ingest_paper
@@ -525,6 +526,69 @@ def test_ingest_sets_bibkey_when_none(tmp_path: Any) -> None:
assert upsert_params["bibkey"] == "AliceBob2023" assert upsert_params["bibkey"] == "AliceBob2023"
def test_ingest_disambiguates_bibkey_on_unique_violation() -> None:
"""A papers.bibkey UNIQUE collision is retried with an a/b/c suffix (audit C-15)."""
paper = _make_paper() # bibkey auto-generates to "AliceBob2023"
mock_conn = MagicMock()
state = {"first": True}
def execute_side_effect(sql: str, params: Any = None) -> MagicMock:
# The first papers INSERT collides on the bibkey; the retry must succeed.
if "INSERT INTO papers" in sql and state["first"]:
state["first"] = False
raise psycopg.errors.UniqueViolation(
'duplicate key value violates unique constraint "papers_bibkey_key"'
)
return MagicMock()
mock_conn.execute.side_effect = execute_side_effect
with (
patch("codex.ingest.openalex.fetch_paper", return_value=paper),
patch("codex.ingest.openalex.fetch_citations", return_value=[]),
patch("codex.ingest.get_embedder", return_value=_fake_embedder()),
patch("codex.ingest.get_conn", side_effect=_make_conn_cm(mock_conn)),
):
ingest_paper(paper.id)
papers_inserts = [
c for c in mock_conn.execute.call_args_list if "INSERT INTO papers" in c[0][0]
]
assert len(papers_inserts) == 2, "expected one retry after the bibkey collision"
assert papers_inserts[0][0][1]["bibkey"] == "AliceBob2023"
assert papers_inserts[1][0][1]["bibkey"] == "AliceBob2023a"
mock_conn.rollback.assert_called_once()
def test_ingest_openalex_collision_raises_clear_error() -> None:
"""An openalex_id UNIQUE collision raises a clear, actionable error (audit C-15).
Unlike a bibkey clash (auto-suffixed), a shared openalex_id means two papers
claim the same OpenAlex work — a real data conflict that must surface, not the
raw psycopg error.
"""
paper = _make_paper()
mock_conn = MagicMock()
def execute_side_effect(sql: str, params: Any = None) -> MagicMock:
if "INSERT INTO papers" in sql:
raise psycopg.errors.UniqueViolation(
'duplicate key value violates unique constraint "papers_openalex_id_key"'
)
return MagicMock()
mock_conn.execute.side_effect = execute_side_effect
with (
patch("codex.ingest.openalex.fetch_paper", return_value=paper),
patch("codex.ingest.openalex.fetch_citations", return_value=[]),
patch("codex.ingest.get_embedder", return_value=_fake_embedder()),
patch("codex.ingest.get_conn", side_effect=_make_conn_cm(mock_conn)),
pytest.raises(ValueError, match="openalex_id"),
):
ingest_paper(paper.id)
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------

View File

@@ -12,6 +12,7 @@ from __future__ import annotations
from unittest.mock import MagicMock, patch from unittest.mock import MagicMock, patch
import pytest import pytest
from pydantic import SecretStr
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# _require_http_token # _require_http_token
@@ -37,7 +38,7 @@ def test_require_http_token_returns_token_when_set() -> None:
from codex.mcp_server import _require_http_token from codex.mcp_server import _require_http_token
mock_settings = MagicMock() mock_settings = MagicMock()
mock_settings.mcp_auth_token = "super-secret-token" mock_settings.mcp_auth_token = SecretStr("super-secret-token")
with patch("codex.config.get_settings", return_value=mock_settings): with patch("codex.config.get_settings", return_value=mock_settings):
result = _require_http_token() result = _require_http_token()
@@ -97,7 +98,7 @@ def test_main_http_with_token_calls_uvicorn() -> None:
mock_settings = MagicMock() mock_settings = MagicMock()
mock_settings.mcp_transport = "http" mock_settings.mcp_transport = "http"
mock_settings.mcp_auth_token = "test-token" mock_settings.mcp_auth_token = SecretStr("test-token")
mock_settings.mcp_host = "127.0.0.1" mock_settings.mcp_host = "127.0.0.1"
mock_settings.mcp_port = 8765 mock_settings.mcp_port = 8765

View File

@@ -10,6 +10,7 @@ from typing import Any
from unittest.mock import MagicMock, patch from unittest.mock import MagicMock, patch
import pytest import pytest
from pydantic import SecretStr
from codex.models import FormulaChunk from codex.models import FormulaChunk
@@ -295,8 +296,8 @@ class TestRouteSelection:
def test_mathpix_selected_when_creds_present(self) -> None: def test_mathpix_selected_when_creds_present(self) -> None:
"""extract_formulas calls MathPix backend when both creds are set.""" """extract_formulas calls MathPix backend when both creds are set."""
mock_settings = MagicMock() mock_settings = MagicMock()
mock_settings.mathpix_app_id = "my_id" mock_settings.mathpix_app_id = SecretStr("my_id")
mock_settings.mathpix_app_key = "my_key" mock_settings.mathpix_app_key = SecretStr("my_key")
mock_settings.pix2tex_fallback = True mock_settings.pix2tex_fallback = True
with ( with (

View File

@@ -8,7 +8,10 @@ from codex.config import Settings, get_settings
def test_settings_defaults() -> None: def test_settings_defaults() -> None:
"""Settings() is constructable without env vars; expected defaults are set.""" """Settings() is constructable without env vars; expected defaults are set."""
s = Settings() s = Settings()
assert s.database_url == "postgresql://researcher:change_me@localhost:5432/papers" assert (
s.database_url.get_secret_value()
== "postgresql://researcher:change_me@localhost:5432/papers"
)
assert s.embedding_model == "BAAI/bge-m3" assert s.embedding_model == "BAAI/bge-m3"
assert s.embedding_dim == 1024 assert s.embedding_dim == 1024
@@ -20,7 +23,7 @@ def test_settings_env_override(monkeypatch: object) -> None:
mp: pytest.MonkeyPatch = monkeypatch # type: ignore[assignment] mp: pytest.MonkeyPatch = monkeypatch # type: ignore[assignment]
mp.setenv("DATABASE_URL", "postgresql://x:y@h:5432/db") mp.setenv("DATABASE_URL", "postgresql://x:y@h:5432/db")
s = Settings() s = Settings()
assert s.database_url == "postgresql://x:y@h:5432/db" assert s.database_url.get_secret_value() == "postgresql://x:y@h:5432/db"
def test_get_settings_is_singleton() -> None: def test_get_settings_is_singleton() -> None:

View File

@@ -33,3 +33,40 @@ def test_apply_schema_executes_sql() -> None:
mock_conn.execute.assert_called_once() mock_conn.execute.assert_called_once()
mock_conn.commit.assert_called_once() mock_conn.commit.assert_called_once()
def test_schema_sql_is_idempotent() -> None:
"""Every CREATE TABLE/INDEX in infra/schema.sql uses IF NOT EXISTS (audit M-1, T-2).
Re-applying schema.sql must be a safe no-op so the migration can run
repeatedly against an existing database.
"""
import re
from pathlib import Path
schema = (Path(__file__).resolve().parents[2] / "infra" / "schema.sql").read_text()
# Drop line comments so commented-out example queries don't count.
code = "\n".join(line.split("--", 1)[0] for line in schema.splitlines())
non_idempotent = re.findall(
r"CREATE\s+(?:UNIQUE\s+)?(?:TABLE|INDEX)\s+(?!IF\s+NOT\s+EXISTS)(\w+)",
code,
re.IGNORECASE,
)
assert not non_idempotent, f"schema.sql has non-idempotent CREATE statements: {non_idempotent}"
def test_migrate_command_reports_privilege_error() -> None:
"""`codex migrate` surfaces InsufficientPrivilege with guidance and exits 1 (M-1)."""
import psycopg
from typer.testing import CliRunner
from codex.cli import app
def boom(*args: object, **kwargs: object) -> object:
raise psycopg.errors.InsufficientPrivilege("must be owner of table chunks")
with patch("psycopg.connect", boom):
result = CliRunner().invoke(app, ["migrate"])
assert result.exit_code == 1
assert "MIGRATION_DATABASE_URL" in result.output

View File

@@ -112,10 +112,35 @@ def test_find_gaps_coverage_map_low_coverage_triggers_llm(
"For an ideal tetrahedron with dihedral angles the hyperbolic volume is " "For an ideal tetrahedron with dihedral angles the hyperbolic volume is "
"V = L(gamma1) + L(gamma2) + L(gamma3). [Springborn2008 #chunk 16]\n" "V = L(gamma1) + L(gamma2) + L(gamma3). [Springborn2008 #chunk 16]\n"
) )
leads = find_gaps(llm=StubLLM(grounded_body)) # Coverage gaps are opt-in (audit R-8): pass the topic explicitly.
leads = find_gaps(llm=StubLLM(grounded_body), topics=["hyperbolic volume formula"])
assert any(lead.kind == "gap" for lead in leads) assert any(lead.kind == "gap" for lead in leads)
def test_find_gaps_no_coverage_gaps_without_explicit_topics(
monkeypatch: pytest.MonkeyPatch,
mock_paper_titles: None,
mock_settings: Path,
) -> None:
"""Without explicit topics, paper titles do NOT auto-generate coverage gaps (R-8)."""
sparse_chunks: list[dict[str, Any]] = [
{
"id": 10,
"paper_id": "springborn-2008",
"ord": 16,
"content": "the hyperbolic volume is V = L(gamma1) + L(gamma2) + L(gamma3)",
"bibkey": "Springborn2008",
},
]
monkeypatch.setattr(
"codex.synthesis._retrieve_chunks",
lambda queries, top_k: [dict(c) for c in sparse_chunks],
)
# No lib_path, no topics → the coverage map must not run despite sparse coverage.
leads = find_gaps(llm=StubLLM("a gap. [Springborn2008 #chunk 16]"))
assert leads == []
def test_find_gaps_explicit_topics_override_default( def test_find_gaps_explicit_topics_override_default(
monkeypatch: pytest.MonkeyPatch, monkeypatch: pytest.MonkeyPatch,
mock_paper_titles: None, mock_paper_titles: None,

View File

@@ -148,7 +148,8 @@ def test_find_gaps_topic_with_no_chunks(
) -> None: ) -> None:
"""When retrieval returns no chunks, a gap lead is emitted directly.""" """When retrieval returns no chunks, a gap lead is emitted directly."""
monkeypatch.setattr("codex.synthesis._retrieve_chunks", lambda queries, top_k: []) monkeypatch.setattr("codex.synthesis._retrieve_chunks", lambda queries, top_k: [])
leads = find_gaps(llm=StubLLM("")) # Coverage gaps are opt-in now (audit R-8): pass an explicit topic.
leads = find_gaps(llm=StubLLM(""), topics=["nonexistent topic"])
assert len(leads) >= 1 assert len(leads) >= 1
assert all(lead.kind == "gap" for lead in leads) assert all(lead.kind == "gap" for lead in leads)
# Direct gap leads carry a validation hint (re-ingest) # Direct gap leads carry a validation hint (re-ingest)

View File

@@ -213,6 +213,47 @@ def test_grounding_guard_last_sentence_grounded_earlier_hallucination_ignored()
) )
def test_grounding_guard_robust_to_trailing_comma() -> None:
"""A faithful 5-word claim grounds despite a trailing comma (audit R-1).
The chunk says "... the volume function V0 is strictly concave on the angle
domain." Before edge-punctuation stripping, the claim token "concave," != the
chunk token "concave", so the only content-5-gram failed to match and the
claim was wrongly marked ungrounded.
"""
claim = Claim(
text="the volume function V0 is strictly concave,",
bibkey="Springborn2008",
locator="chunk 16",
)
result = _run_grounding_guard([claim], FIXTURE_CHUNKS)
assert result[0].grounded is True
def test_grounding_guard_no_substring_bleed_across_tokens() -> None:
"""A 5-gram must match whole tokens, not bleed into a longer chunk word (audit C-6).
The chunk contains 'subset'; a claim whose first token is 'set' must NOT ground
by substring-matching the suffix of 'subset'.
"""
chunks = [
{
"id": 99,
"paper_id": "p",
"ord": 0,
"bibkey": "TestBib",
"content": "the subset conformal map energy functional is minimized here",
}
]
claim = Claim(
text="the set conformal map energy functional",
bibkey="TestBib",
locator="chunk 0",
)
result = _run_grounding_guard([claim], chunks)
assert result[0].grounded is False
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# _inject_cross_refs # _inject_cross_refs
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -274,6 +315,19 @@ def test_inject_cross_refs_skips_inline_code() -> None:
assert "[[circle-packing]]" in result # bare occurrence replaced assert "[[circle-packing]]" in result # bare occurrence replaced
def test_inject_cross_refs_skips_math_spans() -> None:
"""Concept names inside LaTeX math are NOT rewritten to [[slug]] links (audit R-9)."""
concepts = [
Concept(slug="energy", title="Energy", aliases=[]),
FIXTURE_CONCEPT,
]
text = "Prose Energy here. Math $x = Energy^2$ and inline \\(Energy\\) stay literal."
result = _inject_cross_refs(text, concepts, current_slug="lobachevsky-function")
assert "$x = Energy^2$" in result # display/inline $...$ math untouched
assert "\\(Energy\\)" in result # \(...\) math untouched
assert "Prose [[energy]] here." in result # prose occurrence still linked
def test_inject_cross_refs_full_list_even_when_single_concept() -> None: def test_inject_cross_refs_full_list_even_when_single_concept() -> None:
"""Cross-ref injection uses the full concept list, not just the compiled concept.""" """Cross-ref injection uses the full concept list, not just the compiled concept."""
concepts = [ concepts = [