feat(tex): section-aware multi-file .tex ingest (R-C)
Flatten multi-file arXiv LaTeX (\input/\include) in arxiv.fetch_source so the full body is assembled rather than just the primary file's include skeleton, and add section-aware chunking (tex.chunk_sections) so .tex chunks never span a \section boundary and are labelled by their real heading. The ingest .tex path now produces (section, chunk) pairs, quality-gated together so labels stay aligned; the stored 'section' column gains genuine signal instead of mostly 'body' (serves DQ-3 fidelity + audit R-12). Adds scripts/rc_tex_reingest.py to re-ingest arXiv papers from .tex (dry-run by default; replaces chunks). Tests: flatten_inputs, chunk_sections, multi-file fetch_source, section-label ingest, is_arxiv_id. Full suite 365 passed; ruff + mypy clean. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
125
scripts/rc_tex_reingest.py
Normal file
125
scripts/rc_tex_reingest.py
Normal file
@@ -0,0 +1,125 @@
|
||||
"""R-C — re-ingest arXiv papers from LaTeX source (section-aware, higher fidelity).
|
||||
|
||||
For each arXiv paper, download its source via :func:`codex.sources.arxiv.fetch_source`
|
||||
(multi-file ``\\input``/``\\include`` flattened), write it to a temp ``.tex``, and
|
||||
re-ingest through :func:`codex.ingest.ingest_paper`. The ``.tex`` ingest path is
|
||||
section-aware (:func:`codex.parsing.tex.chunk_sections`), so the stored ``section``
|
||||
column is labelled from real ``\\section`` headings instead of mostly ``body`` —
|
||||
serving DQ-3 (fidelity vs the OCR'd ``.txt``) and audit R-12 (weak ``section`` column).
|
||||
|
||||
Only arXiv papers have downloadable source; DOI/journal papers and books are skipped.
|
||||
Re-ingest REPLACES the paper's chunks (DELETE + re-INSERT) — that is the intended R-C
|
||||
improvement (flattened ``.txt`` → clean ``.tex`` chunks). It is idempotent and
|
||||
**dry-run by default**; pass ``--write`` to mutate the live corpus.
|
||||
|
||||
Usage
|
||||
-----
|
||||
# dry-run: list arXiv papers + source availability (no writes)
|
||||
PYTHONPATH=. python scripts/rc_tex_reingest.py
|
||||
|
||||
# re-ingest every arXiv paper from .tex
|
||||
PYTHONPATH=. python scripts/rc_tex_reingest.py --write
|
||||
|
||||
# one paper
|
||||
PYTHONPATH=. python scripts/rc_tex_reingest.py --only 2305.10988 --write
|
||||
|
||||
Needs ``DATABASE_URL`` (Jetson tunnel) as documented in
|
||||
``docs/audit/DATA-QUALITY-2026-06-15.md``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import tempfile
|
||||
|
||||
import psycopg
|
||||
import psycopg.rows
|
||||
|
||||
from codex.db import get_conn
|
||||
from codex.ingest import ingest_paper
|
||||
from codex.sources import arxiv
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def is_arxiv_id(paper_id: str) -> bool:
|
||||
"""Return True for arXiv ids (the only papers with downloadable LaTeX source).
|
||||
|
||||
arXiv ids are modern (``2305.10988``) or legacy (``math/0603097``). DOIs start
|
||||
with ``10.``; OpenAlex W-ids start with ``W`` or an ``https://openalex.org/`` URL.
|
||||
"""
|
||||
p = (paper_id or "").strip()
|
||||
return bool(p) and not p.startswith("10.") and not p.startswith(("W", "https://"))
|
||||
|
||||
|
||||
def _section_dist(conn: psycopg.Connection[psycopg.rows.DictRow], paper_id: str) -> dict[str, int]:
|
||||
"""Return the section-label distribution for a paper's stored chunks."""
|
||||
rows = conn.execute(
|
||||
"SELECT coalesce(section, '(null)') AS s, count(*) AS n "
|
||||
"FROM chunks WHERE paper_id = %s GROUP BY 1 ORDER BY 2 DESC",
|
||||
(paper_id,),
|
||||
).fetchall()
|
||||
return {r["s"]: int(r["n"]) for r in rows}
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--only", default=None, help="restrict to a single arXiv id")
|
||||
parser.add_argument("--limit", type=int, default=None, help="cap number of papers")
|
||||
parser.add_argument(
|
||||
"--write", action="store_true", help="re-ingest into the live DB (default: dry-run)"
|
||||
)
|
||||
args = parser.parse_args(argv)
|
||||
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s")
|
||||
|
||||
with get_conn() as conn:
|
||||
rows = conn.execute("SELECT id FROM papers ORDER BY id").fetchall()
|
||||
targets = [
|
||||
r["id"]
|
||||
for r in rows
|
||||
if is_arxiv_id(r["id"]) and (args.only is None or r["id"] == args.only)
|
||||
]
|
||||
if args.limit is not None:
|
||||
targets = targets[: args.limit]
|
||||
|
||||
logger.info(
|
||||
"arXiv papers to process=%d | mode=%s",
|
||||
len(targets),
|
||||
"WRITE" if args.write else "DRY-RUN",
|
||||
)
|
||||
|
||||
for pid in targets:
|
||||
try:
|
||||
src = arxiv.fetch_source(pid)
|
||||
except Exception as exc:
|
||||
logger.warning("%s: source fetch failed: %s", pid, exc)
|
||||
continue
|
||||
if not src:
|
||||
logger.warning("%s: no LaTeX source available — skipping", pid)
|
||||
continue
|
||||
if not args.write:
|
||||
logger.info("%s: source OK (%d chars)", pid, len(src))
|
||||
continue
|
||||
|
||||
with get_conn() as conn:
|
||||
before = _section_dist(conn, pid)
|
||||
with tempfile.NamedTemporaryFile("w", suffix=".tex", delete=True) as tf:
|
||||
tf.write(src)
|
||||
tf.flush()
|
||||
res = ingest_paper(pid, source_path=tf.name)
|
||||
with get_conn() as conn:
|
||||
after = _section_dist(conn, pid)
|
||||
logger.info(
|
||||
"%s: chunks %d->%d | section before=%s after=%s",
|
||||
pid,
|
||||
sum(before.values()),
|
||||
res.chunks_upserted,
|
||||
before,
|
||||
after,
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user