From ff03443af425460df7796e988ec16d4fa6ce7ba9 Mon Sep 17 00:00:00 2001 From: Tarik Moussa Date: Wed, 17 Jun 2026 00:12:59 +0200 Subject: [PATCH] feat(ingest): recover missing metadata + abstracts from arXiv/S2/Crossref (DQ-2) Three papers lacked abstracts and 1911.00966 was fully degraded (OpenAlex 404 -> empty `Paper(id, title="")` stub). Recovery sources verified, then wired in: - arxiv.fetch_metadata: resolve an arXiv id to a Paper via the Atom export API (authoritative for preprints). Used as an ingest fallback on OpenAlex 404, replacing the empty stub. - crossref.py (new): fetch_abstract(doi), JATS-stripped, Crossref Polite Pool. - semanticscholar.fetch_abstract: fetch just the abstract field. - ingest.py: _recover_abstract() supplements an empty abstract from S2 then Crossref *before* embedding, so the paper gets a real (non-zero) vector. Live DB backfill: 1911.00966 fully recovered from arXiv (bibkey PinkallSpringborn2019, 384-char abstract, its 10 chunks now visible to chunk search); 10.1007/s00454-019-00132-8 abstract from S2 (1186 chars). Book chapter 10.1007/978-3-642-17413-1_7 has no abstract in OpenAlex/S2/Crossref -> documented limit. Corpus now has 1 paper without an abstract. Also documented DQ-5 (not fixed): ingest_paper is not idempotent for DOI papers (openalex returns full-URL id vs the bare canonical id) -> re-ingest trips papers_openalex_id_key. Blocks roadmap R-A/R-C; needs a careful _map_paper fix. Tests: arxiv/crossref/s2 fetchers + ingest recovery paths (342 passing). Co-Authored-By: Claude Opus 4.8 --- codex/ingest.py | 44 +++++++++++-- codex/sources/arxiv.py | 78 +++++++++++++++++++++++ codex/sources/crossref.py | 91 +++++++++++++++++++++++++++ codex/sources/semanticscholar.py | 18 ++++++ docs/audit/DATA-QUALITY-2026-06-15.md | 83 ++++++++++++++++++++---- tests/ingest/test_ingest.py | 74 +++++++++++++++++++++- tests/sources/test_arxiv.py | 60 ++++++++++++++++++ tests/sources/test_crossref.py | 50 +++++++++++++++ tests/sources/test_semanticscholar.py | 43 +++++++++++++ 9 files changed, 521 insertions(+), 20 deletions(-) create mode 100644 codex/sources/crossref.py create mode 100644 tests/sources/test_crossref.py diff --git a/codex/ingest.py b/codex/ingest.py index 5dab3b8..cd950e7 100644 --- a/codex/ingest.py +++ b/codex/ingest.py @@ -13,7 +13,7 @@ from codex.db import get_conn from codex.embed import get_embedder from codex.models import Citation, Paper from codex.quality import classify_section, filter_chunks -from codex.sources import openalex, semanticscholar +from codex.sources import arxiv, crossref, openalex, semanticscholar logger = logging.getLogger(__name__) @@ -71,6 +71,31 @@ def _norm_cited_id(cited_id: str) -> str: return cited_id.lower() if cited_id.startswith("10.") else cited_id +def _recover_abstract(paper: Paper) -> str | None: + """Recover a missing abstract from S2, then Crossref (DQ-2 supplement). + + Tried in order of coverage for this corpus: Semantic Scholar (indexes most + arXiv + journal abstracts), then Crossref (publisher-deposited, DOI only). + Network/parse failures degrade to None rather than aborting the ingest. + """ + s2_id = _s2_id_for(paper.id) + if s2_id is not None: + try: + abstract = semanticscholar.fetch_abstract(s2_id) + if abstract and abstract.strip(): + return abstract.strip() + except Exception: + logger.warning("S2 abstract recovery failed for %s", paper.id, exc_info=True) + if paper.id.startswith("10."): + try: + abstract = crossref.fetch_abstract(paper.id) + if abstract and abstract.strip(): + return abstract.strip() + except Exception: + logger.warning("Crossref abstract recovery failed for %s", paper.id, exc_info=True) + return None + + def _s2_reference_supplement(paper: Paper) -> list[Citation]: """Fetch a paper's references from Semantic Scholar (DQ-1 supplement). @@ -136,16 +161,25 @@ def ingest_paper( len(paper_id) > 4 and paper_id[4:5] == "." and paper_id[:4].isdigit() ) if looks_like_arxiv: - # OpenAlex 404 on an arXiv id → keep a minimal stub. Citation - # recovery from S2 happens in the citation block below (DQ-1 - # supplement), so no probe fetch is needed here. - paper = Paper(id=paper_id, title="") + # OpenAlex 404 on an arXiv id → recover authoritative metadata from + # the arXiv API (DQ-2); fall back to a minimal stub if arXiv also + # has nothing. Citation recovery from S2 happens in the citation + # block below (DQ-1 supplement). + paper = arxiv.fetch_metadata(paper_id) or Paper(id=paper_id, title="") else: raise ValueError(f"Paper not found: {paper_id}") if paper is None: raise ValueError(f"Paper not found: {paper_id}") + # DQ-2: OpenAlex often has the paper but no abstract (it cannot redistribute + # some publishers' abstracts). Supplement from S2, then Crossref, so the + # paper gets a real (non-zero) abstract embedding instead of a zero vector. + if not (paper.abstract or "").strip(): + recovered = _recover_abstract(paper) + if recovered: + paper.abstract = recovered + # Auto-generate bibkey when source has none (D-04). if paper.bibkey is None: paper.bibkey = _make_bibkey(paper) diff --git a/codex/sources/arxiv.py b/codex/sources/arxiv.py index 1de3e38..9bb0303 100644 --- a/codex/sources/arxiv.py +++ b/codex/sources/arxiv.py @@ -1,6 +1,7 @@ """arXiv API client. Provides: +- fetch_metadata: resolve an arXiv ID to a Paper via the Atom export API. - fetch_source: download the .tar.gz source of a paper and extract the primary .tex file. - fetch_pdf_url: return the canonical PDF URL for a given arXiv ID. """ @@ -10,12 +11,89 @@ from __future__ import annotations import io import logging import tarfile +import xml.etree.ElementTree as ET import httpx +from tenacity import retry, retry_if_exception, stop_after_attempt, wait_exponential + +from codex.models import Paper logger = logging.getLogger(__name__) _BASE = "https://arxiv.org" +_EXPORT = "https://export.arxiv.org" +_ATOM = "{http://www.w3.org/2005/Atom}" + + +def _is_retryable(exc: BaseException) -> bool: + if isinstance(exc, httpx.HTTPStatusError): + return exc.response.status_code == 429 or exc.response.status_code >= 500 + return False + + +@retry( + retry=retry_if_exception(_is_retryable), + stop=stop_after_attempt(4), + wait=wait_exponential(min=1, max=20), + before_sleep=lambda rs: logger.warning("arXiv retry %d", rs.attempt_number), +) +def _query(arxiv_id: str) -> httpx.Response: + response = httpx.get( + f"{_EXPORT}/api/query", + params={"id_list": arxiv_id}, + timeout=30, + follow_redirects=True, + ) + response.raise_for_status() + return response + + +def fetch_metadata(arxiv_id: str) -> Paper | None: + """Resolve an arXiv ID to a Paper via the Atom export API. + + Authoritative metadata source for arXiv preprints — used as an ingest + fallback when OpenAlex 404s on an arXiv id (DQ-2). The arXiv id (bare, + e.g. ``"1911.00966"`` or legacy ``"math/0603097"``) becomes ``Paper.id``; + ``openalex_id`` and ``bibkey`` are left for the caller to populate. + + Returns + ------- + Paper | None + Populated Paper (title, authors, year, abstract), or None if arXiv + has no entry for the id. + """ + bare = arxiv_id[len("arxiv:") :] if arxiv_id.lower().startswith("arxiv:") else arxiv_id + try: + response = _query(bare) + except httpx.HTTPError: + logger.warning("arXiv metadata fetch failed for %s", bare, exc_info=True) + return None + try: + root = ET.fromstring(response.text) + except ET.ParseError: + return None + entry = root.find(f"{_ATOM}entry") + if entry is None: + return None + # An id-not-found query still returns a feed but with no . + title = (entry.findtext(f"{_ATOM}title") or "").strip() + if not title: + return None + summary = (entry.findtext(f"{_ATOM}summary") or "").strip() + published = entry.findtext(f"{_ATOM}published") or "" + year = int(published[:4]) if published[:4].isdigit() else None + authors = [ + name.strip() + for a in entry.findall(f"{_ATOM}author") + if (name := a.findtext(f"{_ATOM}name")) and name.strip() + ] + return Paper( + id=bare, + title=" ".join(title.split()), + authors=authors, + year=year, + abstract=" ".join(summary.split()) or None, + ) def fetch_source(arxiv_id: str) -> str | None: diff --git a/codex/sources/crossref.py b/codex/sources/crossref.py new file mode 100644 index 0000000..4727d85 --- /dev/null +++ b/codex/sources/crossref.py @@ -0,0 +1,91 @@ +"""Crossref API client. + +Provides: +- fetch_abstract: retrieve a work's abstract (JATS-stripped) by DOI. + +Crossref is a *third* metadata source after OpenAlex and Semantic Scholar +(DQ-2 / roadmap R-B). Requests use the Polite Pool (mailto query parameter) +and are retried on 429/5xx with exponential back-off via tenacity. +""" + +from __future__ import annotations + +import logging +import re +from typing import Any + +import httpx +from tenacity import retry, retry_if_exception, stop_after_attempt, wait_exponential + +from codex.config import get_settings + +logger = logging.getLogger(__name__) + +_BASE = "https://api.crossref.org" +_TAG_RE = re.compile(r"<[^>]+>") + + +def _is_retryable(exc: BaseException) -> bool: + if isinstance(exc, httpx.HTTPStatusError): + return exc.response.status_code == 429 or exc.response.status_code >= 500 + return False + + +def _polite_params() -> dict[str, str]: + # Reuse the OpenAlex mailto for Crossref's Polite Pool (same address). + mailto = get_settings().openalex_mailto + return {"mailto": mailto} if mailto else {} + + +@retry( + retry=retry_if_exception(_is_retryable), + stop=stop_after_attempt(5), + wait=wait_exponential(min=1, max=30), + before_sleep=lambda rs: logger.warning( + "Crossref retry %d after %s", + rs.attempt_number, + rs.outcome.exception(), # type: ignore[union-attr] + ), +) +def _get(url: str, params: dict[str, str] | None = None) -> httpx.Response: + merged: dict[str, str] = {**_polite_params(), **(params or {})} + response = httpx.get(url, params=merged, timeout=30, follow_redirects=True) + response.raise_for_status() + return response + + +def _strip_jats(abstract: str | None) -> str | None: + """Strip JATS/XML tags from a Crossref abstract and collapse whitespace.""" + if not abstract: + return None + text = _TAG_RE.sub("", abstract) + text = " ".join(text.split()).strip() + return text or None + + +def fetch_abstract(doi: str) -> str | None: + """Fetch a work's abstract from Crossref by DOI. + + Crossref stores abstracts as JATS-tagged markup (````); + tags are stripped before returning. Many works (notably book chapters) + have no abstract deposited — returns None in that case and on 404. + + Parameters + ---------- + doi: + A bare DOI (``"10.1007/s00454-019-00132-8"``). + + Returns + ------- + str | None + The plain-text abstract, or None when absent. + """ + url = f"{_BASE}/works/{doi}" + try: + response = _get(url) + except httpx.HTTPStatusError as exc: + if exc.response.status_code == 404: + return None + raise + message: dict[str, Any] = response.json().get("message", {}) + return _strip_jats(message.get("abstract")) diff --git a/codex/sources/semanticscholar.py b/codex/sources/semanticscholar.py index 0e92268..5862071 100644 --- a/codex/sources/semanticscholar.py +++ b/codex/sources/semanticscholar.py @@ -107,6 +107,24 @@ def fetch_references(paper_id: str) -> list[Citation]: return citations +def fetch_abstract(paper_id: str) -> str | None: + """Fetch just the abstract for a paper from Semantic Scholar. + + Used as an ingest abstract supplement (DQ-2) when OpenAlex has the paper + but no abstract. ``paper_id`` should be namespaced (``arXiv:…`` / ``DOI:…``). + Returns None on 404 or when S2 has no abstract. + """ + url = f"{_BASE_GRAPH}/paper/{paper_id}" + try: + response = _get(url, params={"fields": "abstract"}) + except httpx.HTTPStatusError as exc: + if exc.response.status_code == 404: + return None + raise + abstract = response.json().get("abstract") + return abstract or None + + def fetch_recommendations(paper_id: str, limit: int = 20) -> list[str]: """Fetch recommended paper IDs from Semantic Scholar. diff --git a/docs/audit/DATA-QUALITY-2026-06-15.md b/docs/audit/DATA-QUALITY-2026-06-15.md index b3f87e8..81b8a4c 100644 --- a/docs/audit/DATA-QUALITY-2026-06-15.md +++ b/docs/audit/DATA-QUALITY-2026-06-15.md @@ -2,7 +2,10 @@ **Status:** baseline scan done 2026-06-15. **DQ-1 RESOLVED 2026-06-16** (S2 supplement applied to live DB + wired into ingest). **DQ-4 MEASURED 2026-06-16** — retrieval is -strong after fixing a P0 `codex search paper` crash. **DQ-2/DQ-3 still open.** +strong after fixing a P0 `codex search paper` crash. **DQ-2 RESOLVED 2026-06-16** +(1911.00966 fully recovered from arXiv; s00454 abstract from S2; book chapter is a +documented limit; recovery wired into ingest). **DQ-3 still open. New: DQ-5** +(`ingest_paper` not idempotent for DOI papers — blocks R-A/R-C). **Roadmap R-A..R-E** added 2026-06-16 (free-source acquisition levers — see Roadmap section). **Author:** Audit-Loop (Opus). **Intended reader:** a *cold* session with no memory of the audit conversation — everything needed to continue is in this file. @@ -110,7 +113,32 @@ incomplete, or unrepresentative data. This document is that second axis — a supplement, or (b) documented as an inherent OpenAlex-coverage limit with the graph caveated accordingly. -### DQ-2 — Metadata gaps: 3 no-abstract, 1 fully metadata-less · **MED** +### DQ-2 — Metadata gaps: 3 no-abstract, 1 fully metadata-less · **RESOLVED 2026-06-16** + +**Resolution (what was done):** +- **`1911.00966` fully recovered** from the arXiv Atom API (OpenAlex 404'd on it): + title "A discrete version of Liouville's theorem on conformal maps", + Pinkall & Springborn, 2019, 384-char abstract → bibkey `PinkallSpringborn2019` + auto-generated, abstract embedded (non-zero). It was the worst node in the + corpus — now `@cite`-able, in wiki grounding, and its **10 chunks are visible + to chunk search** (previously filtered out by `bibkey IS NOT NULL`). Citations + were already added in DQ-1 (27 S2 refs). +- **`10.1007/s00454-019-00132-8` abstract recovered from S2** (1186 chars) and + re-embedded (targeted `UPDATE`, not full re-ingest — see DQ-5). +- **`10.1007/978-3-642-17413-1_7` (book chapter): documented inherent limit.** + No abstract in OpenAlex, S2, **or Crossref** (verified). Left as the single + remaining zero-vector paper; user may add an abstract by hand. +- **Corpus now: 1 paper without abstract** (the book chapter), down from 3+1. + Zero-vector search pollution (DQ-4 secondary finding) reduced from 2 tail + fillers to 1. +- **Wired into ingest** so this self-heals: OpenAlex-404 on an arXiv id now + recovers metadata via `arxiv.fetch_metadata` (new); an empty abstract is + supplemented via `_recover_abstract` → S2 then Crossref (new + `codex/sources/crossref.py`). New `semanticscholar.fetch_abstract`. +- **Files touched:** `codex/sources/arxiv.py`, `codex/sources/crossref.py` (new), + `codex/sources/semanticscholar.py`, `codex/ingest.py`, + tests. Full suite: 342. + +**Original finding (for context):** - **Measured:** - **No abstract (3):** `10.1007/978-3-642-17413-1_7`, `10.1007/s00454-019-00132-8`, `1911.00966`. These get a **zero-vector** @@ -187,6 +215,30 @@ is fine (monotonic); only the absolute score label is misleading. - **Acceptance:** a short relevance table (query → top-5 → on/off-topic) good enough to trust the KB for lookups, or a list of failure modes to fix. +### DQ-5 — `ingest_paper` is NOT idempotent for DOI papers · **HIGH (blocks R-A/R-C)** · found 2026-06-16 +- **Symptom:** re-ingesting an existing DOI paper crashes with + `UniqueViolation: papers_openalex_id_key`. Surfaced trying to re-ingest + `10.1007/s00454-019-00132-8` during the DQ-2 backfill. +- **Root cause:** `openalex._map_paper` sets `Paper.id = data["doi"]`, which is the + **full URL** `https://doi.org/10.1007/…`. The stored canonical id is the **bare** + `10.1007/…` (post the M-1 canonical-id migration). So `INSERT … ON CONFLICT (id)` + finds no id match, attempts a fresh INSERT, and trips the *separate* unique + constraint on `openalex_id` (already held by the canonical-id row). +- **Confirmed:** `openalex.fetch_paper("10.1007/s00454-019-00132-8").id == + "https://doi.org/10.1007/s00454-019-00132-8"` (≠ stored bare id). +- **Impact:** any re-ingest of a DOI paper aborts. This **blocks roadmap R-A + (GROBID on PDFs) and R-C (.tex ingest)** — both re-ingest existing papers. Also + implies *new* DOI ingests store full-URL ids, inconsistent with the bare-id + corpus (latent split-identity risk). +- **Proposed fix (handle with care — M-1 incident territory):** normalize the DOI + in `_map_paper` to the canonical bare, lower-cased form (strip + `https://doi.org/`), matching what the M-1 migration produced. Add a re-ingest + idempotency test over a DOI paper. Did **not** fix inline this session — the + canonical-id derivation is exactly where the M-1 migration incident occurred, so + it needs its own focused change + a live re-ingest check. +- **Workaround used for DQ-2:** targeted `UPDATE … SET abstract, abstract_emb` + instead of full re-ingest. + --- ## Roadmap — data-acquisition levers (post-audit, all free sources) @@ -200,6 +252,8 @@ Paywall/uni-login was explicitly considered and **deferred** (see R-E). Each ite is self-contained so a cold session can pick it up. ### R-A — Run GROBID reference extraction on arXiv PDFs · **HIGH lever, LOW-MED effort** +- **⚠ Blocked by DQ-5:** re-ingesting existing DOI papers currently crashes + (`papers_openalex_id_key`). Fix DQ-5 first, or the GROBID re-ingest aborts. - **What:** the whole corpus was ingested from `.txt` (`PAPERS_DIR=…/papers/txt`), so the GROBID PDF path never ran. Re-ingest with a PDF `source_path` per paper so `codex.parsing.grobid.extract_references` parses each bibliography. @@ -268,18 +322,21 @@ is self-contained so a cold session can pick it up. coverage; remedy wired into ingest). 2. ~~**DQ-4**~~ — **DONE 2026-06-16** (P0 `search paper` crash fixed; relevance verified strong; surfaced the zero-vector pollution that motivates DQ-2). -3. **DQ-2** (NEXT — recover `1911.00966`; backfill 3 missing abstracts so they - stop polluting paper-level search with zero-vectors), then **DQ-3** (chunk-vs- - source `.txt` fidelity). -4. **Roadmap levers** (see section above), best sequenced after DQ-2/DQ-3: - **R-A** (GROBID on arXiv PDFs — closes the last 2 zero-citation theses, highest - ROI), then **R-B** (Crossref source — also recovers DQ-2 abstracts), **R-C** - (`.tex` ingest — serves DQ-3), **R-D** (F-15 coverage warning, quick). - **R-E** (paywall) is deferred. +3. ~~**DQ-2**~~ — **DONE 2026-06-16** (1911.00966 fully recovered; s00454 abstract + from S2; book chapter documented; recovery wired into ingest). +4. **DQ-5** (NEXT — `ingest_paper` not idempotent for DOI papers; **blocks R-A/R-C**; + careful — M-1 canonical-id territory), then **DQ-3** (chunk-vs-source fidelity). +5. **Roadmap levers** (see section above): **R-A** (GROBID on arXiv PDFs — closes + the last 2 zero-citation theses; needs DQ-5 first), **R-B** (Crossref refs — + abstract half already shipped in DQ-2), **R-C** (`.tex` ingest — serves DQ-3; + needs DQ-5 first), **R-D** (F-15 coverage warning, quick). **R-E** (paywall) + deferred. -DQ-1..DQ-4 are *data*/assessment work; the roadmap items R-A..R-D add **new code** -(re-ingest scripts, a Crossref source module, a `.tex` path, a graph warning) on -top of the already-remediated pipeline (see the AUDIT-* docs and PRs #12–#14). +DQ-1..DQ-4 are *data*/assessment work; DQ-5 + roadmap R-A..R-D add **new code** +(canonical-id fix, re-ingest scripts, Crossref references, a `.tex` path, a graph +warning) on top of the already-remediated pipeline (see the AUDIT-* docs and PRs +#12–#14). The DQ-1/DQ-2 ingest wiring + Crossref/arXiv abstract recovery already +landed this session. --- diff --git a/tests/ingest/test_ingest.py b/tests/ingest/test_ingest.py index 6aad09b..5729466 100644 --- a/tests/ingest/test_ingest.py +++ b/tests/ingest/test_ingest.py @@ -250,7 +250,7 @@ def test_ingest_paper_idempotent() -> None: def test_ingest_paper_arxiv_s2_fallback() -> None: - """If OpenAlex returns None for an arXiv ID, S2 fallback produces a stub Paper.""" + """OpenAlex None + arXiv has nothing → stub Paper; S2 references still consulted.""" paper_id = "2301.07041" mock_conn = MagicMock() mock_conn.execute = MagicMock() @@ -259,17 +259,84 @@ def test_ingest_paper_arxiv_s2_fallback() -> None: with ( patch("codex.ingest.openalex.fetch_paper", return_value=None), + # arXiv + abstract sources also empty → genuine stub fallback path. + patch("codex.ingest.arxiv.fetch_metadata", return_value=None) as mock_arxiv, + patch("codex.ingest.semanticscholar.fetch_abstract", return_value=None), + patch("codex.ingest.crossref.fetch_abstract", return_value=None), patch("codex.ingest.semanticscholar.fetch_references", return_value=[]) as mock_s2, patch("codex.ingest.get_embedder", return_value=_fake_embedder()), patch("codex.ingest.get_conn", side_effect=_make_conn_cm(mock_conn)), ): result = ingest_paper(paper_id) - # S2 was consulted as fallback + # arXiv metadata recovery attempted; S2 references consulted as fallback. + mock_arxiv.assert_called_once_with(paper_id) mock_s2.assert_called() assert result.paper_id == paper_id +def test_ingest_paper_openalex_404_recovers_from_arxiv() -> None: + """DQ-2: OpenAlex 404 on an arXiv id → authoritative metadata from the arXiv API, + bibkey auto-generated, abstract embedded (no zero vector).""" + recovered = Paper( + id="1911.00966", + title="A discrete version of Liouville's theorem on conformal maps", + authors=["Ulrich Pinkall", "Boris Springborn"], + year=2019, + abstract="Liouville's theorem says that in dimension greater than two...", + ) + fake_emb = _fake_embedder() + mock_conn = MagicMock() + mock_conn.execute = MagicMock() + mock_conn.executemany = MagicMock() + mock_conn.commit = MagicMock() + + with ( + patch("codex.ingest.openalex.fetch_paper", return_value=None), + patch("codex.ingest.arxiv.fetch_metadata", return_value=recovered) as mock_arxiv, + patch("codex.ingest.semanticscholar.fetch_references", return_value=[]), + patch("codex.ingest.get_embedder", return_value=fake_emb), + patch("codex.ingest.get_conn", side_effect=_make_conn_cm(mock_conn)), + ): + result = ingest_paper("1911.00966") + + mock_arxiv.assert_called_once_with("1911.00966") + assert result.paper_id == "1911.00966" + params: dict[str, Any] = mock_conn.execute.call_args_list[0][0][1] + assert params["title"].startswith("A discrete version") + assert params["bibkey"] == "PinkallSpringborn2019" # generated from recovered authors+year + # abstract present → real embedding computed, not a zero vector + fake_emb.encode_dense.assert_called_once() + + +def test_ingest_paper_empty_abstract_recovered_from_s2() -> None: + """DQ-2: OpenAlex has the paper but no abstract → supplemented from S2 before embedding.""" + paper = _make_paper(abstract=None) # has openalex_id, no abstract + fake_emb = _fake_embedder() + mock_conn = MagicMock() + mock_conn.execute = MagicMock() + mock_conn.executemany = MagicMock() + mock_conn.commit = MagicMock() + + with ( + patch("codex.ingest.openalex.fetch_paper", return_value=paper), + patch("codex.ingest.openalex.fetch_citations", return_value=[]), + patch("codex.ingest.semanticscholar.fetch_references", return_value=[]), + patch( + "codex.ingest.semanticscholar.fetch_abstract", + return_value="Recovered abstract text.", + ) as mock_abs, + patch("codex.ingest.get_embedder", return_value=fake_emb), + patch("codex.ingest.get_conn", side_effect=_make_conn_cm(mock_conn)), + ): + ingest_paper(paper.id) + + mock_abs.assert_called_once() + params: dict[str, Any] = mock_conn.execute.call_args_list[0][0][1] + assert params["abstract"] == "Recovered abstract text." + fake_emb.encode_dense.assert_called_once_with(["Recovered abstract text."]) + + def test_ingest_paper_openalex_empty_uses_s2_supplement() -> None: """DQ-1: OpenAlex returns an empty reference list → ingest supplements from S2, rewriting citing_id to paper.id and lowercasing DOI cited-ids.""" @@ -325,6 +392,9 @@ def test_ingest_paper_no_abstract_uses_zero_vector() -> None: with ( patch("codex.ingest.openalex.fetch_paper", return_value=paper), patch("codex.ingest.openalex.fetch_citations", return_value=[]), + # No abstract recoverable from any source (DQ-2) → genuine zero-vector case. + patch("codex.ingest.semanticscholar.fetch_abstract", return_value=None), + patch("codex.ingest.crossref.fetch_abstract", return_value=None), patch("codex.ingest.get_embedder", return_value=fake_emb), patch("codex.ingest.get_conn", side_effect=_make_conn_cm(mock_conn)), ): diff --git a/tests/sources/test_arxiv.py b/tests/sources/test_arxiv.py index 1c08500..c4ceb60 100644 --- a/tests/sources/test_arxiv.py +++ b/tests/sources/test_arxiv.py @@ -138,3 +138,63 @@ def test_fetch_pdf_url_pure_computation() -> None: url2 = arxiv.fetch_pdf_url("1234.56789") assert url2 == "https://arxiv.org/pdf/1234.56789.pdf" + + +# --------------------------------------------------------------------------- +# fetch_metadata — parse the Atom feed into a Paper (DQ-2) +# --------------------------------------------------------------------------- + +_ATOM_FEED = """ + + + A discrete version of Liouville's theorem on conformal maps + Liouville's theorem says that in dimension greater than two, + all conformal maps are Moebius transformations. + 2019-11-03T00:00:00Z + Ulrich Pinkall + Boris Springborn + +""" + +_ATOM_EMPTY = """ +ArXiv Query""" + + +def test_fetch_metadata_parses_entry(monkeypatch: pytest.MonkeyPatch) -> None: + """A valid Atom feed yields a populated Paper.""" + + def mock_get(url: str, **kwargs: object) -> httpx.Response: + return httpx.Response(200, text=_ATOM_FEED, request=httpx.Request("GET", url)) + + monkeypatch.setattr(httpx, "get", mock_get) + + paper = arxiv.fetch_metadata("1911.00966") + assert paper is not None + assert paper.id == "1911.00966" + assert paper.title.startswith("A discrete version of Liouville") + assert paper.authors == ["Ulrich Pinkall", "Boris Springborn"] + assert paper.year == 2019 + assert paper.abstract and "conformal maps" in paper.abstract + assert paper.openalex_id is None + + +def test_fetch_metadata_no_entry_returns_none(monkeypatch: pytest.MonkeyPatch) -> None: + """A feed with no (id not found) returns None.""" + + def mock_get(url: str, **kwargs: object) -> httpx.Response: + return httpx.Response(200, text=_ATOM_EMPTY, request=httpx.Request("GET", url)) + + monkeypatch.setattr(httpx, "get", mock_get) + assert arxiv.fetch_metadata("0000.00000") is None + + +def test_fetch_metadata_strips_arxiv_prefix(monkeypatch: pytest.MonkeyPatch) -> None: + """An ``arXiv:`` prefix is stripped so Paper.id is the bare id.""" + + def mock_get(url: str, **kwargs: object) -> httpx.Response: + return httpx.Response(200, text=_ATOM_FEED, request=httpx.Request("GET", url)) + + monkeypatch.setattr(httpx, "get", mock_get) + paper = arxiv.fetch_metadata("arXiv:1911.00966") + assert paper is not None + assert paper.id == "1911.00966" diff --git a/tests/sources/test_crossref.py b/tests/sources/test_crossref.py new file mode 100644 index 0000000..f88d007 --- /dev/null +++ b/tests/sources/test_crossref.py @@ -0,0 +1,50 @@ +"""Tests for codex.sources.crossref.""" + +from __future__ import annotations + +import httpx +import pytest + +from codex.sources import crossref + + +def test_fetch_abstract_strips_jats(monkeypatch: pytest.MonkeyPatch) -> None: + """A JATS-tagged abstract is returned as plain text with tags removed.""" + body = { + "message": { + "abstract": "We prove a theorem about " + "polyhedra." + } + } + + def mock_get(url: str, params: dict[str, object] | None = None) -> httpx.Response: + return httpx.Response(200, json=body) + + monkeypatch.setattr(crossref, "_get", mock_get) + + result = crossref.fetch_abstract("10.1007/s00454-019-00132-8") + assert result == "We prove a theorem about polyhedra." + + +def test_fetch_abstract_absent_returns_none(monkeypatch: pytest.MonkeyPatch) -> None: + """A work with no abstract field (e.g. a book chapter) returns None.""" + + def mock_get(url: str, params: dict[str, object] | None = None) -> httpx.Response: + return httpx.Response(200, json={"message": {"title": ["Some Chapter"]}}) + + monkeypatch.setattr(crossref, "_get", mock_get) + assert crossref.fetch_abstract("10.1007/978-3-642-17413-1_7") is None + + +def test_fetch_abstract_404_returns_none(monkeypatch: pytest.MonkeyPatch) -> None: + """A 404 returns None rather than raising.""" + + def mock_get(url: str, params: dict[str, object] | None = None) -> httpx.Response: + raise httpx.HTTPStatusError( + "404", + request=httpx.Request("GET", url), + response=httpx.Response(404, request=httpx.Request("GET", url)), + ) + + monkeypatch.setattr(crossref, "_get", mock_get) + assert crossref.fetch_abstract("10.0000/nonexistent") is None diff --git a/tests/sources/test_semanticscholar.py b/tests/sources/test_semanticscholar.py index 42066d6..a1fb726 100644 --- a/tests/sources/test_semanticscholar.py +++ b/tests/sources/test_semanticscholar.py @@ -124,3 +124,46 @@ def test_fetch_recommendations_404_returns_empty( result = semanticscholar.fetch_recommendations("nonexistent") assert result == [] + + +# --------------------------------------------------------------------------- +# fetch_references — {"data": null} guard (DQ-1 regression) +# --------------------------------------------------------------------------- + + +def test_fetch_references_null_data_returns_empty(monkeypatch: pytest.MonkeyPatch) -> None: + """S2 returns HTTP 200 with ``{"data": null}`` for a known paper that has no + parsed references; this must yield [] rather than raising TypeError.""" + + def mock_get(url: str, params: dict[str, object] | None = None) -> httpx.Response: + return httpx.Response(200, json={"data": None, "citingPaperInfo": {}}) + + monkeypatch.setattr(semanticscholar, "_get", mock_get) + assert semanticscholar.fetch_references("DOI:10.14279/depositonce-5415") == [] + + +# --------------------------------------------------------------------------- +# fetch_abstract (DQ-2) +# --------------------------------------------------------------------------- + + +def test_fetch_abstract_returns_text(monkeypatch: pytest.MonkeyPatch) -> None: + """fetch_abstract returns the abstract string when present.""" + + def mock_get(url: str, params: dict[str, object] | None = None) -> httpx.Response: + return httpx.Response(200, json={"abstract": "We study ideal hyperbolic polyhedra."}) + + monkeypatch.setattr(semanticscholar, "_get", mock_get) + assert semanticscholar.fetch_abstract("DOI:10.1007/s00454-019-00132-8") == ( + "We study ideal hyperbolic polyhedra." + ) + + +def test_fetch_abstract_null_returns_none(monkeypatch: pytest.MonkeyPatch) -> None: + """A null/absent abstract yields None.""" + + def mock_get(url: str, params: dict[str, object] | None = None) -> httpx.Response: + return httpx.Response(200, json={"abstract": None}) + + monkeypatch.setattr(semanticscholar, "_get", mock_get) + assert semanticscholar.fetch_abstract("arXiv:1234.5678") is None