fix(review): address PR #16 code-review findings
#1 ingest: normalize the pinned caller id (_norm_cited_id) so a non-bare DOI caller cannot store a URL-form papers.id that defeats DQ-5 idempotency / the startswith(10.) recovery gates. #2 quality.section_label: collapse to a controlled bucket only for an EXACT canonical heading; descriptive titles ('Abstract Nonsense...') keep their real title instead of being mislabelled. #4 ra_grobid_backfill: release the read connection before the slow GROBID network loop, fresh connection for the write (no idle-in-transaction across the loop over the flaky tunnel). #5/#10 tex: flatten_inputs strips unresolved input/include at the depth cap (no literal leak on cycles); _norm_texkey strips only a single leading ./ . #6/#7 arxiv.fetch_source: keep non-.tex members resolvable for input; pick primary on an UN-commented documentclass line. #13 is_arxiv_id: also exclude http:// and arXiv-DOI forms. Tests added/updated for each. Left as deliberate decisions: #3 (pre-section text drop is pre-existing in extract_sections; abstract stored separately), #8 (script normalizer is intentionally self-contained, already documented), #9/#11/#12 (no current trigger / tightening the gzip heuristic would reject valid old LaTeX like documentstyle / title truncation is by design on a write-only column). ruff + mypy clean. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -97,6 +97,16 @@ def fetch_metadata(arxiv_id: str) -> Paper | None:
|
||||
)
|
||||
|
||||
|
||||
def _has_documentclass(content: str) -> bool:
|
||||
"""True if *content* has an UN-commented ``\\documentclass`` line.
|
||||
|
||||
A plain ``"\\documentclass" in content`` substring test also matches a
|
||||
commented-out ``%\\documentclass`` (common in arXiv preambles); requiring the
|
||||
command to start its line avoids selecting the wrong primary file.
|
||||
"""
|
||||
return any(line.lstrip().startswith("\\documentclass") for line in content.splitlines())
|
||||
|
||||
|
||||
def fetch_source(arxiv_id: str) -> str | None:
|
||||
"""Download and extract the primary LaTeX source for an arXiv paper.
|
||||
|
||||
@@ -128,29 +138,38 @@ def fetch_source(arxiv_id: str) -> str | None:
|
||||
from codex.parsing.tex import flatten_inputs
|
||||
|
||||
raw = response.content
|
||||
# Image/font/binary members are never \input'd as text; skip them so the source
|
||||
# map stays text-only (decode-with-replace would otherwise store garbled bytes).
|
||||
skip_ext = (".png", ".jpg", ".jpeg", ".gif", ".bmp", ".pdf", ".eps", ".ps", ".ttf", ".otf")
|
||||
try:
|
||||
with tarfile.open(fileobj=io.BytesIO(raw), mode="r:gz") as tf:
|
||||
files: dict[str, str] = {}
|
||||
primary_name: str | None = None
|
||||
for member in tf.getmembers():
|
||||
if not member.name.endswith(".tex"):
|
||||
if not member.isfile() or member.name.lower().endswith(skip_ext):
|
||||
continue
|
||||
f = tf.extractfile(member)
|
||||
if f is None:
|
||||
continue
|
||||
content = f.read().decode("utf-8", errors="replace")
|
||||
files[member.name] = content
|
||||
# Prefer the \documentclass file as the primary document.
|
||||
if primary_name is None and "\\documentclass" in content:
|
||||
# Keep ALL text members resolvable so \input of a non-.tex include
|
||||
# (e.g. a .def or extension-less body file) is not silently dropped (R-C).
|
||||
files[member.name] = f.read().decode("utf-8", errors="replace")
|
||||
# Primary doc = a .tex with an UN-commented \documentclass.
|
||||
if (
|
||||
primary_name is None
|
||||
and member.name.endswith(".tex")
|
||||
and _has_documentclass(files[member.name])
|
||||
):
|
||||
primary_name = member.name
|
||||
|
||||
if not files:
|
||||
tex_files = {k: v for k, v in files.items() if k.endswith(".tex")}
|
||||
if not tex_files:
|
||||
logger.debug("No .tex files found in arXiv source for %s", arxiv_id)
|
||||
return None
|
||||
|
||||
# No \documentclass anywhere → fall back to the largest .tex.
|
||||
if primary_name is None:
|
||||
primary_name = max(files, key=lambda k: len(files[k]))
|
||||
primary_name = max(tex_files, key=lambda k: len(tex_files[k]))
|
||||
|
||||
# Inline \input/\include so multi-file projects yield the full body,
|
||||
# not just the primary file's skeleton of include directives (R-C).
|
||||
|
||||
Reference in New Issue
Block a user