Data-quality roadmap: R-A–R-F + DQ-5 fixes (citations 590→1075, section signal 10%→95%) #16
@@ -12,7 +12,7 @@ from codex.config import get_settings
|
|||||||
from codex.db import get_conn
|
from codex.db import get_conn
|
||||||
from codex.embed import get_embedder
|
from codex.embed import get_embedder
|
||||||
from codex.models import Citation, Paper
|
from codex.models import Citation, Paper
|
||||||
from codex.quality import classify_section, is_quality_chunk
|
from codex.quality import is_quality_chunk, section_label
|
||||||
from codex.sources import arxiv, crossref, openalex, semanticscholar
|
from codex.sources import arxiv, crossref, openalex, semanticscholar
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
@@ -312,10 +312,11 @@ def ingest_paper(
|
|||||||
ord_idx,
|
ord_idx,
|
||||||
content,
|
content,
|
||||||
chunk_embeddings[ord_idx].tolist(),
|
chunk_embeddings[ord_idx].tolist(),
|
||||||
# .tex: classify from the section heading (prepended to the
|
# .tex: label from the real \section heading — keeps the
|
||||||
# snippet) so the label reflects the real section; otherwise
|
# controlled bucket when it matches, else stores the cleaned
|
||||||
# fall back to content-only classification.
|
# title (R-F); .txt/.pdf (title=None) classify by content.
|
||||||
classify_section(f"{title}. {content}" if title else content),
|
# NB: `section` semantics now differ by source (write-only col).
|
||||||
|
section_label(title, content),
|
||||||
)
|
)
|
||||||
for ord_idx, (title, content) in enumerate(kept_pairs)
|
for ord_idx, (title, content) in enumerate(kept_pairs)
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -110,6 +110,42 @@ def classify_section(text: str) -> str:
|
|||||||
return "body"
|
return "body"
|
||||||
|
|
||||||
|
|
||||||
|
def _clean_title(title: str) -> str:
|
||||||
|
"""Normalize a LaTeX ``\\section`` title for use as a section label.
|
||||||
|
|
||||||
|
Strips LaTeX commands (``\\emph`` …) and ``{}$``, drops leading section
|
||||||
|
numbering (``3.2 ``), collapses whitespace, lower-cases, and truncates to
|
||||||
|
60 chars so the stored ``section`` value is a clean, comparable string.
|
||||||
|
"""
|
||||||
|
t = re.sub(r"\\[a-zA-Z]+\*?", " ", title) # LaTeX commands
|
||||||
|
t = re.sub(r"[{}$]", "", t) # braces / math delimiters
|
||||||
|
t = re.sub(r"^\s*\d+(?:\.\d+)*\.?\s*", "", t) # leading numbering "3.2 "
|
||||||
|
return " ".join(t.split()).strip().lower()[:60]
|
||||||
|
|
||||||
|
|
||||||
|
def section_label(title: str | None, content: str) -> str:
|
||||||
|
"""Label a chunk's section, preferring the real ``\\section`` heading (R-F).
|
||||||
|
|
||||||
|
For section-aware ``.tex`` ingest *title* is the real heading: if it maps to a
|
||||||
|
controlled bucket via :func:`classify_section` (intro / theorem / proof /
|
||||||
|
abstract / bibliography) that bucket is kept (cross-source consistency);
|
||||||
|
otherwise the cleaned title itself is stored (e.g. ``"preliminaries"``,
|
||||||
|
``"rigidity"``) — far more signal than collapsing everything to ``body``.
|
||||||
|
|
||||||
|
When *title* is None / blank (the ``.pdf`` / ``.txt`` / ``run_quality_pass``
|
||||||
|
paths) this falls back to content-based :func:`classify_section` — unchanged
|
||||||
|
behaviour. Safe to store free-text titles because the ``section`` column is
|
||||||
|
write-only (no consumer filters on the controlled vocabulary).
|
||||||
|
"""
|
||||||
|
if not title or not title.strip():
|
||||||
|
return classify_section(content)
|
||||||
|
cleaned = _clean_title(title)
|
||||||
|
if not cleaned:
|
||||||
|
return classify_section(content)
|
||||||
|
bucket = classify_section(cleaned)
|
||||||
|
return bucket if bucket != "body" else cleaned
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Batch filter
|
# Batch filter
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|||||||
@@ -128,6 +128,7 @@ def test_ingest_paper_source_tex(tmp_path: Any) -> None:
|
|||||||
section_pairs = [
|
section_pairs = [
|
||||||
("Introduction", "Body of the introduction goes here."),
|
("Introduction", "Body of the introduction goes here."),
|
||||||
("Proof of the Main Theorem", "Body of the proof goes here."),
|
("Proof of the Main Theorem", "Body of the proof goes here."),
|
||||||
|
("3.2 Discrete Conformal Maps", "Body about discrete conformal maps."),
|
||||||
]
|
]
|
||||||
|
|
||||||
with (
|
with (
|
||||||
@@ -151,9 +152,11 @@ def test_ingest_paper_source_tex(tmp_path: Any) -> None:
|
|||||||
chunk_sql: str = chunk_insert_call[0][0]
|
chunk_sql: str = chunk_insert_call[0][0]
|
||||||
assert "INSERT INTO chunks" in chunk_sql
|
assert "INSERT INTO chunks" in chunk_sql
|
||||||
|
|
||||||
# Section label comes from the heading: "Introduction" → intro, "Proof of…" → proof.
|
# R-F: heading → controlled bucket where it maps ("Introduction"→intro,
|
||||||
|
# "Proof of…"→proof), else the cleaned real title ("discrete conformal maps")
|
||||||
|
# instead of collapsing to "body".
|
||||||
chunk_rows = chunk_insert_call[0][1]
|
chunk_rows = chunk_insert_call[0][1]
|
||||||
assert [row[4] for row in chunk_rows] == ["intro", "proof"]
|
assert [row[4] for row in chunk_rows] == ["intro", "proof", "discrete conformal maps"]
|
||||||
assert result.chunks_upserted == len(section_pairs)
|
assert result.chunks_upserted == len(section_pairs)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -6,10 +6,12 @@ from unittest.mock import MagicMock
|
|||||||
|
|
||||||
from codex.quality import (
|
from codex.quality import (
|
||||||
_bib_score,
|
_bib_score,
|
||||||
|
_clean_title,
|
||||||
classify_section,
|
classify_section,
|
||||||
filter_chunks,
|
filter_chunks,
|
||||||
is_quality_chunk,
|
is_quality_chunk,
|
||||||
run_quality_pass,
|
run_quality_pass,
|
||||||
|
section_label,
|
||||||
)
|
)
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -243,3 +245,36 @@ class TestRunQualityPass:
|
|||||||
kwargs = update_calls[0][0][1]
|
kwargs = update_calls[0][0][1]
|
||||||
assert kwargs["section"] == "abstract"
|
assert kwargs["section"] == "abstract"
|
||||||
assert kwargs["id"] == 5
|
assert kwargs["id"] == 5
|
||||||
|
|
||||||
|
|
||||||
|
class TestCleanTitle:
|
||||||
|
def test_strips_leading_numbering(self):
|
||||||
|
assert _clean_title("3.2 Variational Principles") == "variational principles"
|
||||||
|
|
||||||
|
def test_strips_latex_and_braces(self):
|
||||||
|
assert _clean_title(r"The \emph{Discrete} Laplacian") == "the discrete laplacian"
|
||||||
|
|
||||||
|
def test_collapses_and_lowercases(self):
|
||||||
|
assert _clean_title(" Rigidity Results ") == "rigidity results"
|
||||||
|
|
||||||
|
|
||||||
|
class TestSectionLabel:
|
||||||
|
def test_title_maps_to_controlled_bucket(self):
|
||||||
|
# Headings matching the controlled vocab keep the canonical bucket.
|
||||||
|
assert section_label("Introduction", "body text") == "intro"
|
||||||
|
assert section_label("Proof of Theorem 3", "body text") == "proof"
|
||||||
|
assert section_label("References", "body text") == "bibliography"
|
||||||
|
|
||||||
|
def test_descriptive_title_stored_verbatim(self):
|
||||||
|
# R-F win: a non-vocab heading is stored as its cleaned title, not "body".
|
||||||
|
assert section_label("Preliminaries", "body text") == "preliminaries"
|
||||||
|
assert section_label("3.1 Discrete Conformal Maps", "x") == "discrete conformal maps"
|
||||||
|
|
||||||
|
def test_none_or_blank_title_falls_back_to_content(self):
|
||||||
|
# .pdf/.txt path: unchanged content-based classification.
|
||||||
|
assert section_label(None, "Theorem 1. Let X be ...") == "theorem"
|
||||||
|
assert section_label("", "ordinary body prose here") == "body"
|
||||||
|
|
||||||
|
def test_title_cleaning_to_empty_falls_back_to_content(self):
|
||||||
|
# A pure-numbering heading cleans to "" → classify by content instead.
|
||||||
|
assert section_label("3.2", "Lemma 2. Suppose ...") == "theorem"
|
||||||
|
|||||||
Reference in New Issue
Block a user