Files
codex-py/codex/sources/semanticscholar.py
Tarik Moussa 51dc74470c feat(ingest): supplement citations from Semantic Scholar when OpenAlex is empty (DQ-1)
12/29 papers had zero citations: OpenAlex indexes arXiv preprints and theses
without a parsed reference list (verified referenced_works_count=0 server-side
for all 11 with an openalex_id). Not an ingest bug — a source-coverage limit.

- ingest.py: when OpenAlex returns no references, fall back to a Semantic
  Scholar reference supplement (_s2_reference_supplement); citing_id rewritten
  to canonical papers.id, DOI cited-ids lowercased. Removed a now-redundant
  discarded S2 probe in the OpenAlex-404 arXiv branch.
- semanticscholar.py: fix TypeError on S2's HTTP-200 {"data": null} no-refs
  responses (.get("data", []) returns None when the key is present-but-null).
- tests: regression test for the OpenAlex-empty -> S2 path.
- Live DB backfilled idempotently: +330 citations (590->920), zero-out-edge
  papers 12->2, citing coverage 17->27/29. Only the 2 TU-Berlin depositonce
  theses remain (no references in any source).
- docs: DATA-QUALITY-2026-06-15.md — DQ-1 resolution, DQ-4 result, and a
  free-source acquisition roadmap (R-A..R-E).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-16 23:51:48 +02:00

137 lines
4.2 KiB
Python

"""Semantic Scholar API client.
Provides:
- fetch_references: retrieve references for a paper as Citation dataclasses.
- fetch_recommendations: retrieve recommended paper IDs.
Rate-limited to ≤1 req/s (per-request floor via monotonic clock).
Retried on 429/5xx with exponential back-off via tenacity.
"""
from __future__ import annotations
import logging
import threading
import time
from typing import Any
import httpx
from tenacity import retry, retry_if_exception, stop_after_attempt, wait_exponential
from codex.models import Citation
logger = logging.getLogger(__name__)
_BASE_GRAPH = "https://api.semanticscholar.org/graph/v1"
_BASE_RECS = "https://api.semanticscholar.org/recommendations/v1"
# Per-request rate limit: ≤1 req/s without an API key.
_rate_lock = threading.Lock()
_last_request_time: float = 0.0
_MIN_INTERVAL = 1.0
def _rate_limit() -> None:
global _last_request_time
with _rate_lock:
now = time.monotonic()
wait = _MIN_INTERVAL - (now - _last_request_time)
if wait > 0:
time.sleep(wait)
_last_request_time = time.monotonic()
def _is_retryable(exc: BaseException) -> bool:
if isinstance(exc, httpx.HTTPStatusError):
return exc.response.status_code == 429 or exc.response.status_code >= 500
return False
@retry(
retry=retry_if_exception(_is_retryable),
stop=stop_after_attempt(5),
wait=wait_exponential(min=1, max=30),
before_sleep=lambda rs: logger.warning(
"SemanticScholar retry %d after %s",
rs.attempt_number,
rs.outcome.exception(), # type: ignore[union-attr]
),
)
def _get(url: str, params: dict[str, Any] | None = None) -> httpx.Response:
_rate_limit()
response = httpx.get(url, params=params, timeout=30)
response.raise_for_status()
return response
def fetch_references(paper_id: str) -> list[Citation]:
"""Fetch references for a paper from Semantic Scholar.
Parameters
----------
paper_id:
Semantic Scholar paper ID (or ``arXiv:…`` / ``DOI:…`` prefixed ID).
Returns
-------
list[Citation]
One Citation per reference, with optional context snippet.
"""
url = f"{_BASE_GRAPH}/paper/{paper_id}/references"
params: dict[str, Any] = {"fields": "externalIds,contexts"}
try:
response = _get(url, params=params)
except httpx.HTTPStatusError as exc:
if exc.response.status_code == 404:
return []
raise
data = response.json()
# S2 returns HTTP 200 with ``{"data": null}`` for a known paper that has no
# parsed reference list (e.g. some theses). ``.get("data", [])`` would yield
# the explicit ``None`` (the default only applies when the key is absent),
# so iterate over a guaranteed list instead.
raw_refs: list[dict[str, Any]] = data.get("data") or []
citations: list[Citation] = []
for entry in raw_refs:
cited_paper: dict[str, Any] = entry.get("citedPaper", {})
external_ids: dict[str, str] = cited_paper.get("externalIds") or {}
contexts: list[str] = entry.get("contexts", [])
context: str | None = contexts[0] if contexts else None
cited_id: str = (
external_ids.get("DOI") or external_ids.get("ArXiv") or cited_paper.get("paperId") or ""
)
if cited_id:
citations.append(Citation(citing_id=paper_id, cited_id=cited_id, context=context))
return citations
def fetch_recommendations(paper_id: str, limit: int = 20) -> list[str]:
"""Fetch recommended paper IDs from Semantic Scholar.
Parameters
----------
paper_id:
Semantic Scholar paper ID.
limit:
Maximum number of recommendations to return.
Returns
-------
list[str]
List of recommended paper IDs.
"""
url = f"{_BASE_RECS}/papers/forpaper/{paper_id}"
params: dict[str, Any] = {"limit": limit}
try:
response = _get(url, params=params)
except httpx.HTTPStatusError as exc:
if exc.response.status_code == 404:
return []
raise
data = response.json()
recommended: list[dict[str, Any]] = data.get("recommendedPapers", [])
return [p["paperId"] for p in recommended if p.get("paperId")]