feat(sources): OpenAlex, SemanticScholar, arXiv API clients
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
140
tests/sources/test_arxiv.py
Normal file
140
tests/sources/test_arxiv.py
Normal file
@@ -0,0 +1,140 @@
|
||||
"""Tests for codex.sources.arxiv."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import tarfile
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from codex.sources import arxiv
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _make_tar_gz(files: dict[str, bytes]) -> bytes:
|
||||
"""Build an in-memory .tar.gz archive from a dict of {name: content}."""
|
||||
buf = io.BytesIO()
|
||||
with tarfile.open(fileobj=buf, mode="w:gz") as tf:
|
||||
for name, content in files.items():
|
||||
info = tarfile.TarInfo(name=name)
|
||||
info.size = len(content)
|
||||
tf.addfile(info, io.BytesIO(content))
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# fetch_source — success: in-memory tar.gz with one .tex
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_fetch_source_returns_latex(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""fetch_source extracts and returns the LaTeX content from a .tar.gz."""
|
||||
latex_content = b"\\documentclass{article}\n\\begin{document}\nHello world\n\\end{document}"
|
||||
tar_bytes = _make_tar_gz({"main.tex": latex_content})
|
||||
|
||||
def mock_get(
|
||||
url: str,
|
||||
*,
|
||||
timeout: int = 60,
|
||||
follow_redirects: bool = True,
|
||||
) -> httpx.Response:
|
||||
return httpx.Response(200, content=tar_bytes)
|
||||
|
||||
monkeypatch.setattr(httpx, "get", mock_get)
|
||||
|
||||
result = arxiv.fetch_source("2301.07041")
|
||||
|
||||
assert result is not None
|
||||
assert "\\documentclass" in result
|
||||
assert "Hello world" in result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# fetch_source — 404 → None
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_fetch_source_404_returns_none(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""A 404 response should return None."""
|
||||
|
||||
def mock_get(
|
||||
url: str,
|
||||
*,
|
||||
timeout: int = 60,
|
||||
follow_redirects: bool = True,
|
||||
) -> httpx.Response:
|
||||
return httpx.Response(404)
|
||||
|
||||
monkeypatch.setattr(httpx, "get", mock_get)
|
||||
|
||||
result = arxiv.fetch_source("nonexistent-id")
|
||||
assert result is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# fetch_source — no .tex in archive → None
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_fetch_source_no_tex_returns_none(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""If the archive contains no .tex files, return None."""
|
||||
tar_bytes = _make_tar_gz({"README.md": b"# Paper\nNo LaTeX here."})
|
||||
|
||||
def mock_get(
|
||||
url: str,
|
||||
*,
|
||||
timeout: int = 60,
|
||||
follow_redirects: bool = True,
|
||||
) -> httpx.Response:
|
||||
return httpx.Response(200, content=tar_bytes)
|
||||
|
||||
monkeypatch.setattr(httpx, "get", mock_get)
|
||||
|
||||
result = arxiv.fetch_source("2301.07041")
|
||||
assert result is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# fetch_source — fallback to largest .tex when no \documentclass
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_fetch_source_fallback_to_largest_tex(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""When no file has \\documentclass, the largest .tex is returned."""
|
||||
small_tex = b"% small file\n\\section{Intro}"
|
||||
large_tex = b"% large file\n" + b"x" * 500
|
||||
|
||||
tar_bytes = _make_tar_gz({"small.tex": small_tex, "large.tex": large_tex})
|
||||
|
||||
def mock_get(
|
||||
url: str,
|
||||
*,
|
||||
timeout: int = 60,
|
||||
follow_redirects: bool = True,
|
||||
) -> httpx.Response:
|
||||
return httpx.Response(200, content=tar_bytes)
|
||||
|
||||
monkeypatch.setattr(httpx, "get", mock_get)
|
||||
|
||||
result = arxiv.fetch_source("2301.07041")
|
||||
|
||||
assert result is not None
|
||||
assert "large file" in result
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# fetch_pdf_url — pure computation, no mock needed
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_fetch_pdf_url_pure_computation() -> None:
|
||||
"""fetch_pdf_url returns the correct URL without making any HTTP request."""
|
||||
url = arxiv.fetch_pdf_url("2301.07041")
|
||||
assert url == "https://arxiv.org/pdf/2301.07041.pdf"
|
||||
|
||||
url2 = arxiv.fetch_pdf_url("1234.56789")
|
||||
assert url2 == "https://arxiv.org/pdf/1234.56789.pdf"
|
||||
Reference in New Issue
Block a user