diff --git a/codex/cli.py b/codex/cli.py index ede4f8d..ac0c14e 100644 --- a/codex/cli.py +++ b/codex/cli.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +from datetime import UTC, datetime from typing import Optional import typer @@ -10,27 +11,48 @@ import typer app = typer.Typer(help="codex — personal knowledge base for scientific papers.") discover_app = typer.Typer(help="Discovery queries over the citation graph.") prov_app = typer.Typer(help="Provenance: @cite scan, code_links, bib export.") +search_app = typer.Typer( + help="Search commands: paper abstracts (default) or formulas (formula subcommand).", + invoke_without_command=True, +) +wiki_app = typer.Typer(help="Wiki-compile: grounded concept pages over the RAG substrate.") app.add_typer(discover_app, name="discover") app.add_typer(prov_app, name="provenance") +app.add_typer(search_app, name="search") +app.add_typer(wiki_app, name="wiki") @app.command() def ingest( paper_id: str = typer.Argument(..., help="arXiv ID, DOI, or OpenAlex W-ID"), source: Optional[str] = typer.Option(None, "--source", "-s", help="Path to .tex or .pdf"), # noqa: UP045 + rich: bool = typer.Option( # noqa: FBT002 + False, "--rich", help="Also extract formulas and figures (PDF only)." + ), ) -> None: """Ingest a paper into the knowledge base.""" from codex.ingest import ingest_paper - result = ingest_paper(paper_id, source_path=source) - typer.echo( + result = ingest_paper(paper_id, source_path=source, rich=rich) + msg = ( f"Ingested {result.paper_id}: {result.chunks_upserted} chunks, " f"{result.citations_upserted} citations" ) + if rich: + msg += f", {result.formulas_upserted} formulas, {result.figures_upserted} figures" + typer.echo(msg) -@app.command() -def search( +@search_app.callback(invoke_without_command=True) +def search_callback(ctx: typer.Context) -> None: + """Search commands for papers and formulas.""" + if ctx.invoked_subcommand is None: + typer.echo(ctx.get_help()) + raise typer.Exit(0) + + +@search_app.command("paper") +def search_paper( query: str = typer.Argument(..., help="Natural-language search query"), limit: int = typer.Option(10, "--limit", "-n", help="Number of results"), ) -> None: @@ -155,3 +177,167 @@ def ask(question: str = typer.Argument(..., help="Question to answer")) -> None: """Ask a question (RAG — not yet implemented).""" typer.echo("ask: not yet implemented", err=True) raise typer.Exit(1) + + +# --------------------------------------------------------------------------- +# F-09: Specialised search commands +# --------------------------------------------------------------------------- + + +@search_app.command("formula") +def search_formula( + query: str = typer.Argument(..., help="LaTeX snippet or natural-language description"), + limit: int = typer.Option(10, "--limit", "-n", help="Number of results"), +) -> None: + """Search for mathematical formulas by LaTeX content or surrounding context. + + Performs a full-text search over ``formulas.raw_latex`` and + ``formulas.context``, ranked by relevance. + """ + from codex.db import get_conn + + _fts = "to_tsvector('english', coalesce(f.raw_latex, '') || ' ' || coalesce(f.context, ''))" + with get_conn() as conn: + rows = conn.execute( + f""" + SELECT f.id, f.paper_id, f.page, f.raw_latex, f.context, f.eq_label, + ts_rank({_fts}, plainto_tsquery('english', %(query)s)) AS rank + FROM formulas f + WHERE {_fts} @@ plainto_tsquery('english', %(query)s) + ORDER BY rank DESC + LIMIT %(limit)s + """, + {"query": query, "limit": limit}, + ).fetchall() + + if not rows: + typer.echo("No formula results found.") + return + + for row in rows: + label = f" [{row['eq_label']}]" if row["eq_label"] else "" + typer.echo( + f"[rank={row['rank']:.3f}] {row['paper_id']} p.{row['page']}{label}\n" + f" LaTeX: {row['raw_latex']}\n" + f" Context: {row['context'][:120] if row['context'] else ''}\n" + ) + + +# --------------------------------------------------------------------------- +# F-12: Wiki command group +# --------------------------------------------------------------------------- + + +@wiki_app.command("compile") +def wiki_compile( + concept: Optional[str] = typer.Option( # noqa: UP045 + None, "--concept", help="Compile only this concept slug." + ), + all_concepts: bool = typer.Option( + False, "--all", help="Force full recompile (ignore change detection)." + ), + top_k: int = typer.Option(0, "--top-k", help="Override config.wiki_top_k (0 = use config)."), + output_dir: Optional[str] = typer.Option( # noqa: UP045 + None, "--output-dir", help="Override config.wiki_dir." + ), +) -> None: + """Compile concept pages from retrieved chunks via local LLM. + + By default only recompiles concepts whose source chunks have changed + (hash-based incremental). Use --all to force full recompile. + """ + from codex.wiki import compile_all + + report = compile_all( + changed_only=not all_concepts, + top_k=top_k if top_k > 0 else None, + concept_filter=concept, + output_dir=output_dir, + ) + + if report.compiled: + typer.echo(f"Compiled: {', '.join(report.compiled)}") + if report.skipped: + typer.echo(f"Skipped (unchanged): {', '.join(report.skipped)}") + if report.ungrounded: + typer.echo(f"⚠ Ungrounded claims: {len(report.ungrounded)}", err=True) + for slug, text in report.ungrounded: + typer.echo(f" [{slug}] {text[:100]}", err=True) + + +@wiki_app.command("list") +def wiki_list( + output_dir: Optional[str] = typer.Option( # noqa: UP045 + None, "--output-dir", help="Override config.wiki_dir." + ), +) -> None: + """List compiled concept pages with freshness information.""" + from pathlib import Path + + from codex.config import get_settings + + settings = get_settings() + wiki_dir = Path(output_dir or settings.wiki_dir) + state_path = wiki_dir / ".compile-state.json" + + state: dict[str, str] = {} + if state_path.exists(): + try: + state = json.loads(state_path.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError): + state = {} + + pages = sorted(wiki_dir.glob("*.md")) + if not pages: + typer.echo("No compiled pages found.") + return + + for page in pages: + if page.name in ("index.md", "log.md"): + continue + slug = page.stem + hash_val = state.get(slug, "—")[:8] + # Count claims and ungrounded in the page markdown + content = page.read_text(encoding="utf-8") + n_claims = content.count("[") + n_ungrounded = content.count("⚠") + mtime = datetime.fromtimestamp(page.stat().st_mtime, tz=UTC).strftime( + "%Y-%m-%d %H:%M" + ) + typer.echo( + f"{slug:40s} last={mtime} hash={hash_val} claims≈{n_claims} ⚠={n_ungrounded}" + ) + + +@wiki_app.command("check") +def wiki_check( + output_dir: Optional[str] = typer.Option( # noqa: UP045 + None, "--output-dir", help="Override config.wiki_dir." + ), +) -> None: + """Check all compiled pages for ungrounded claims. + + Exits with code 0 if all claims are grounded, 1 if any are ungrounded. + Suitable for CI pipelines. + """ + from pathlib import Path + + from codex.config import get_settings + + settings = get_settings() + wiki_dir = Path(output_dir or settings.wiki_dir) + + pages = sorted(wiki_dir.glob("*.md")) + found_ungrounded = False + for page in pages: + if page.name in ("index.md", "log.md"): + continue + content = page.read_text(encoding="utf-8") + if "⚠" in content: + typer.echo(f"⚠ Ungrounded claim(s) in: {page.name}", err=True) + found_ungrounded = True + + if found_ungrounded: + raise typer.Exit(1) + else: + typer.echo("All claims grounded.") diff --git a/codex/config.py b/codex/config.py index a6f8bd7..35e6a2a 100644 --- a/codex/config.py +++ b/codex/config.py @@ -84,6 +84,80 @@ class Settings(BaseSettings): ), ) + # ------------------------------------------------------------------ + # F-09 Rich Parsing — MathPix + pix2tex + figures + # ------------------------------------------------------------------ + mathpix_app_id: str | None = Field( + default=None, + description=( + "MathPix application ID for the MathPix API " + "(env var: MATHPIX_APP_ID). Optional — pix2tex is used as fallback." + ), + ) + + mathpix_app_key: str | None = Field( + default=None, + description=( + "MathPix application key for the MathPix API " + "(env var: MATHPIX_APP_KEY). Optional — pix2tex is used as fallback." + ), + ) + + pix2tex_fallback: bool = Field( + default=True, + description=( + "Enable pix2tex LatexOCR as the local fallback for formula extraction " + "when MathPix credentials are not set. Disable only for testing." + ), + ) + + figures_dir: str = Field( + default="figures/", + description=( + "Directory where extracted figure images (PNG) are written. " + "Relative paths are resolved from the current working directory." + ), + ) + + # ------------------------------------------------------------------ + # F-12 Wiki-Compile + # ------------------------------------------------------------------ + wiki_dir: str = Field( + default="wiki/", + description=( + "Directory where compiled wiki pages are written. " + "Relative paths are resolved from the current working directory." + ), + ) + + wiki_llm_model: str = Field( + default="qwen2.5:7b", + description=( + "Ollama model name used for wiki synthesis. " + "Must be available at the configured Ollama endpoint (WIKI_LLM_URL or " + "OLLAMA_BASE_URL). Default: qwen2.5:7b (qwen-light profile on Jetson)." + ), + ) + + wiki_llm_url: str | None = Field( + default=None, + description=( + "Ollama base URL for wiki synthesis. " + "When None, falls back to OLLAMA_BASE_URL " + "(http://192.168.178.103:11434 for the Jetson). " + "Set WIKI_LLM_URL to override." + ), + ) + + wiki_top_k: int = Field( + default=12, + gt=0, + description=( + "Number of top chunks retrieved per concept for wiki synthesis. " + "Higher values improve recall at the cost of a larger LLM prompt." + ), + ) + @lru_cache(maxsize=1) def get_settings() -> Settings: