feat(tex): section-aware multi-file .tex ingest (R-C)
Flatten multi-file arXiv LaTeX (\input/\include) in arxiv.fetch_source so the full body is assembled rather than just the primary file's include skeleton, and add section-aware chunking (tex.chunk_sections) so .tex chunks never span a \section boundary and are labelled by their real heading. The ingest .tex path now produces (section, chunk) pairs, quality-gated together so labels stay aligned; the stored 'section' column gains genuine signal instead of mostly 'body' (serves DQ-3 fidelity + audit R-12). Adds scripts/rc_tex_reingest.py to re-ingest arXiv papers from .tex (dry-run by default; replaces chunks). Tests: flatten_inputs, chunk_sections, multi-file fetch_source, section-label ingest, is_arxiv_id. Full suite 365 passed; ruff + mypy clean. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -101,8 +101,9 @@ def fetch_source(arxiv_id: str) -> str | None:
|
||||
|
||||
Downloads the .tar.gz source bundle from ``https://arxiv.org/src/{arxiv_id}``,
|
||||
locates the primary .tex file (preferring any file containing ``\\documentclass``,
|
||||
falling back to the largest .tex by size), and returns its contents as a UTF-8
|
||||
string.
|
||||
falling back to the largest .tex by size), and inlines its ``\\input``/
|
||||
``\\include`` directives from the other archive members so multi-file projects
|
||||
return the full document body, not just the primary file's include skeleton.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
@@ -126,33 +127,36 @@ def fetch_source(arxiv_id: str) -> str | None:
|
||||
if response.status_code != 200:
|
||||
response.raise_for_status()
|
||||
|
||||
from codex.parsing.tex import flatten_inputs
|
||||
|
||||
raw = response.content
|
||||
try:
|
||||
with tarfile.open(fileobj=io.BytesIO(raw), mode="r:gz") as tf:
|
||||
tex_members = [m for m in tf.getmembers() if m.name.endswith(".tex")]
|
||||
if not tex_members:
|
||||
logger.debug("No .tex files found in arXiv source for %s", arxiv_id)
|
||||
return None
|
||||
|
||||
# Prefer the file containing \documentclass (primary document)
|
||||
primary: tarfile.TarInfo | None = None
|
||||
for member in tex_members:
|
||||
files: dict[str, str] = {}
|
||||
primary_name: str | None = None
|
||||
for member in tf.getmembers():
|
||||
if not member.name.endswith(".tex"):
|
||||
continue
|
||||
f = tf.extractfile(member)
|
||||
if f is None:
|
||||
continue
|
||||
content_bytes = f.read()
|
||||
if b"\\documentclass" in content_bytes:
|
||||
primary = member
|
||||
# Decode and return immediately — first match wins
|
||||
return content_bytes.decode("utf-8", errors="replace")
|
||||
content = f.read().decode("utf-8", errors="replace")
|
||||
files[member.name] = content
|
||||
# Prefer the \documentclass file as the primary document.
|
||||
if primary_name is None and "\\documentclass" in content:
|
||||
primary_name = member.name
|
||||
|
||||
if primary is None:
|
||||
# Fallback: largest .tex by size
|
||||
largest = max(tex_members, key=lambda m: m.size)
|
||||
f = tf.extractfile(largest)
|
||||
if f is None:
|
||||
return None
|
||||
return f.read().decode("utf-8", errors="replace")
|
||||
if not files:
|
||||
logger.debug("No .tex files found in arXiv source for %s", arxiv_id)
|
||||
return None
|
||||
|
||||
# No \documentclass anywhere → fall back to the largest .tex.
|
||||
if primary_name is None:
|
||||
primary_name = max(files, key=lambda k: len(files[k]))
|
||||
|
||||
# Inline \input/\include so multi-file projects yield the full body,
|
||||
# not just the primary file's skeleton of include directives (R-C).
|
||||
return flatten_inputs(files[primary_name], files)
|
||||
except tarfile.TarError as exc:
|
||||
logger.warning("Failed to open tar archive for %s: %s", arxiv_id, exc)
|
||||
return None
|
||||
|
||||
Reference in New Issue
Block a user