#1 ingest: normalize the pinned caller id (_norm_cited_id) so a non-bare DOI caller cannot store a URL-form papers.id that defeats DQ-5 idempotency / the startswith(10.) recovery gates. #2 quality.section_label: collapse to a controlled bucket only for an EXACT canonical heading; descriptive titles ('Abstract Nonsense...') keep their real title instead of being mislabelled. #4 ra_grobid_backfill: release the read connection before the slow GROBID network loop, fresh connection for the write (no idle-in-transaction across the loop over the flaky tunnel). #5/#10 tex: flatten_inputs strips unresolved input/include at the depth cap (no literal leak on cycles); _norm_texkey strips only a single leading ./ . #6/#7 arxiv.fetch_source: keep non-.tex members resolvable for input; pick primary on an UN-commented documentclass line. #13 is_arxiv_id: also exclude http:// and arXiv-DOI forms. Tests added/updated for each. Left as deliberate decisions: #3 (pre-section text drop is pre-existing in extract_sections; abstract stored separately), #8 (script normalizer is intentionally self-contained, already documented), #9/#11/#12 (no current trigger / tightening the gzip heuristic would reject valid old LaTeX like documentstyle / title truncation is by design on a write-only column). ruff + mypy clean. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
249 lines
9.2 KiB
Python
249 lines
9.2 KiB
Python
"""Chunk quality filtering and section classification (F-16).
|
|
|
|
Three independent quality signals are applied at ingest time:
|
|
|
|
1. **Length** — structural; no content required.
|
|
2. **Alpha-ratio** — OCR artefacts have high non-alpha character density.
|
|
3. **Bib-score** — DOI + "et al." + (YYYY) patterns co-occur almost only
|
|
in reference list entries.
|
|
|
|
Section classification is rule-based (no LLM) and runs on the first 200
|
|
characters of each chunk. The retroactive ``run_quality_pass`` function
|
|
can be called from the CLI to back-fill existing chunks.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import re
|
|
from typing import TYPE_CHECKING, Any
|
|
|
|
if TYPE_CHECKING:
|
|
from codex.config import Settings
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Section classification patterns (checked in order; first match wins)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_SECTION_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
|
|
("abstract", re.compile(r"^\s*(abstract|zusammenfassung|r[eé]sum[eé])\b", re.I)),
|
|
("intro", re.compile(r"^\s*(introduction|einleitung|1\.\s)", re.I)),
|
|
("theorem", re.compile(r"^\s*(theorem|lemma|proposition|corollary|definition)\b", re.I)),
|
|
("proof", re.compile(r"^\s*(proof\b|beweis\b|proof\s+of\b)", re.I)),
|
|
("bibliography", re.compile(r"^\s*(references|bibliography|bibliographie|literatur)\b", re.I)),
|
|
]
|
|
|
|
_DOI_RE = re.compile(r"10\.\d{4,}/\S+")
|
|
_ET_AL_RE = re.compile(r"\bet\s+al\b", re.I)
|
|
_YEAR_BRACKETS_RE = re.compile(r"\(\d{4}\)")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Bibliography heuristic score
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _bib_score(text: str) -> float:
|
|
"""Return a [0..1] heuristic for how bibliography-like a chunk is.
|
|
|
|
Three signals contribute:
|
|
* DOI occurrences (weight 3)
|
|
* "et al." occurrences (weight 2)
|
|
* year-in-brackets occurrences (weight 1)
|
|
|
|
The raw signal is normalised against a rough word-count proxy so that
|
|
long chunks with occasional references don't get flagged.
|
|
"""
|
|
if not text:
|
|
return 0.0
|
|
word_count = max(len(text.split()), 1)
|
|
doi_hits = len(_DOI_RE.findall(text))
|
|
etal_hits = len(_ET_AL_RE.findall(text))
|
|
year_hits = len(_YEAR_BRACKETS_RE.findall(text))
|
|
raw = (doi_hits * 3 + etal_hits * 2 + year_hits) / max(word_count / 10, 1)
|
|
return min(raw, 1.0)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Quality predicate
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def is_quality_chunk(text: str, *, settings: Settings) -> bool:
|
|
"""Return True when *text* passes all three quality thresholds.
|
|
|
|
Checks (all configurable via :class:`codex.config.Settings`):
|
|
|
|
1. ``chunk_min_chars`` — character count floor.
|
|
2. ``chunk_min_alpha_ratio`` — minimum fraction of alphabetic chars.
|
|
3. ``chunk_max_bib_score`` — bibliography heuristic ceiling.
|
|
"""
|
|
if len(text) < settings.chunk_min_chars:
|
|
return False
|
|
alpha_ratio = sum(c.isalpha() for c in text) / max(len(text), 1)
|
|
if alpha_ratio < settings.chunk_min_alpha_ratio:
|
|
return False
|
|
return _bib_score(text) <= settings.chunk_max_bib_score
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Section classification
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def classify_section(text: str) -> str:
|
|
"""Classify a chunk's section using rule-based regex matching.
|
|
|
|
Inspects only the first 200 characters. Returns one of:
|
|
``abstract``, ``intro``, ``theorem``, ``proof``, ``bibliography``, ``body``.
|
|
"""
|
|
snippet = text[:200]
|
|
for section_name, pattern in _SECTION_PATTERNS:
|
|
if pattern.search(snippet):
|
|
return section_name
|
|
# Fallback: bibliography by DOI density even without a header
|
|
if _bib_score(text) > 0.5:
|
|
return "bibliography"
|
|
return "body"
|
|
|
|
|
|
def _clean_title(title: str) -> str:
|
|
"""Normalize a LaTeX ``\\section`` title for use as a section label.
|
|
|
|
Strips LaTeX word-commands (``\\emph`` …), accent/escape macros (``\\"o`` →
|
|
``o``, ``\\&``), ``~`` ties and ``{}$``, drops leading section numbering
|
|
(``3.2 ``), collapses whitespace, lower-cases, and truncates to 60 chars at a
|
|
word boundary so the stored ``section`` value is a clean, comparable string.
|
|
"""
|
|
t = title.replace("~", " ") # LaTeX non-breaking tie → space
|
|
t = re.sub(r"\\[a-zA-Z]+\*?", " ", t) # word-commands: \emph, \mathcal …
|
|
t = re.sub(r"\\[^a-zA-Z]", "", t) # accent/escape macros: \"o → o, \', \`, \&
|
|
t = re.sub(r"[{}$]", "", t) # braces / math delimiters
|
|
t = re.sub(r"^\s*\d+(?:\.\d+)*\.?\s*", "", t) # leading numbering "3.2 "
|
|
t = " ".join(t.split()).strip().lower()
|
|
return t[:60].rsplit(" ", 1)[0] if len(t) > 60 else t
|
|
|
|
|
|
# A heading collapses to a controlled bucket only when it IS one of these canonical
|
|
# section words — NOT when it merely starts with one. Prefix-matching via
|
|
# classify_section mislabels real sections ("Abstract Nonsense and Categories" →
|
|
# abstract, "References Architecture" → bibliography); an exact map avoids that while
|
|
# keeping cross-source consistency for the bare headings.
|
|
_SECTION_TITLE_BUCKETS = {
|
|
"abstract": "abstract",
|
|
"zusammenfassung": "abstract",
|
|
"introduction": "intro",
|
|
"einleitung": "intro",
|
|
"proof": "proof",
|
|
"beweis": "proof",
|
|
"references": "bibliography",
|
|
"bibliography": "bibliography",
|
|
"bibliographie": "bibliography",
|
|
"literatur": "bibliography",
|
|
}
|
|
|
|
|
|
def section_label(title: str | None, content: str) -> str:
|
|
"""Label a chunk's section, preferring the real ``\\section`` heading (R-F).
|
|
|
|
For section-aware ``.tex`` ingest *title* is the real heading: a bare canonical
|
|
heading ("Introduction", "References", "Proof") maps to its controlled bucket
|
|
(cross-source consistency); any other heading is stored as its cleaned real
|
|
title (e.g. ``"preliminaries"``, ``"introduction to operator algebras"``) — far
|
|
more signal than collapsing to ``body`` AND without the prefix-match mislabels
|
|
that ``classify_section`` would produce on descriptive titles.
|
|
|
|
When *title* is None / blank (the ``.pdf`` / ``.txt`` / ``run_quality_pass``
|
|
paths) this falls back to content-based :func:`classify_section` — unchanged
|
|
behaviour. Safe to store free-text titles because the ``section`` column is
|
|
write-only (no consumer filters on the controlled vocabulary).
|
|
"""
|
|
if not title or not title.strip():
|
|
return classify_section(content)
|
|
cleaned = _clean_title(title)
|
|
if not cleaned:
|
|
return classify_section(content)
|
|
return _SECTION_TITLE_BUCKETS.get(cleaned, cleaned)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Batch filter
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def filter_chunks(chunks: list[str], *, settings: Settings) -> list[str]:
|
|
"""Return only the chunks that pass all quality filters.
|
|
|
|
Logs the keep ratio at DEBUG level.
|
|
"""
|
|
kept = [c for c in chunks if is_quality_chunk(c, settings=settings)]
|
|
logger.debug(
|
|
"Quality filter: %d/%d chunks kept (%.0f%%)",
|
|
len(kept),
|
|
len(chunks),
|
|
100 * len(kept) / max(len(chunks), 1),
|
|
)
|
|
return kept
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Retroactive DB pass
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def run_quality_pass(
|
|
*,
|
|
paper_id: str | None = None,
|
|
conn: Any,
|
|
settings: Settings,
|
|
) -> dict[str, int]:
|
|
"""Apply quality filter + section classification to existing chunks in DB.
|
|
|
|
For each chunk:
|
|
* Fails quality → DELETE.
|
|
* Passes quality → UPDATE ``section`` with :func:`classify_section`.
|
|
|
|
Parameters
|
|
----------
|
|
paper_id:
|
|
When provided, restrict the pass to chunks for this paper only.
|
|
conn:
|
|
Open psycopg connection (dict-row factory assumed).
|
|
settings:
|
|
Application settings for quality thresholds.
|
|
|
|
Returns
|
|
-------
|
|
dict with keys ``kept``, ``removed``, ``tagged`` (= kept).
|
|
"""
|
|
if paper_id is not None:
|
|
rows = conn.execute(
|
|
"SELECT id, content FROM chunks WHERE paper_id = %(pid)s",
|
|
{"pid": paper_id},
|
|
).fetchall()
|
|
else:
|
|
rows = conn.execute("SELECT id, content FROM chunks").fetchall()
|
|
|
|
kept = 0
|
|
removed = 0
|
|
for row in rows:
|
|
chunk_id = row["id"]
|
|
content = row["content"]
|
|
if not is_quality_chunk(content, settings=settings):
|
|
conn.execute("DELETE FROM chunks WHERE id = %(id)s", {"id": chunk_id})
|
|
removed += 1
|
|
else:
|
|
section = classify_section(content)
|
|
conn.execute(
|
|
"UPDATE chunks SET section = %(section)s WHERE id = %(id)s",
|
|
{"section": section, "id": chunk_id},
|
|
)
|
|
kept += 1
|
|
|
|
conn.commit()
|
|
logger.info("Quality pass done: %d kept, %d removed, %d section-tagged", kept, removed, kept)
|
|
return {"kept": kept, "removed": removed, "tagged": kept}
|