Files
codex-py/tests/parsing/test_grobid.py
Tarik Moussa d1ffea1c1c fix(parsing): case-insensitive arXiv idno match; fix tmp_path types; add extract_structure test
Review-Gate finding: GROBID emits type="arXiv" (camel-case), not "arxiv".
XPath literal match silently returned empty strings for all real responses.
Fix: iterate idno elements and compare .lower() == "arxiv".
Test fixture updated to reflect real GROBID output (type="arXiv").

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-04 23:47:18 +02:00

162 lines
5.2 KiB
Python

"""Tests for codex.parsing.grobid."""
from __future__ import annotations
from pathlib import Path
from unittest.mock import MagicMock, patch
from codex.parsing.grobid import extract_references, extract_structure
# Real GROBID TEI output uses type="arXiv" (camel-case), not "arxiv".
_TEI_XML = """\
<?xml version="1.0" encoding="UTF-8"?>
<TEI xmlns="http://www.tei-c.org/ns/1.0">
<text>
<back>
<div type="references">
<listBibl>
<biblStruct>
<analytic>
<title level="a">Deep Learning for NLP</title>
<author>
<persName>
<forename>Jane</forename>
<surname>Doe</surname>
</persName>
</author>
</analytic>
<monogr>
<imprint>
<date type="published" when="2021-06"/>
</imprint>
</monogr>
<idno type="DOI">10.1234/nlp.2021</idno>
<idno type="arXiv">2101.00001</idno>
</biblStruct>
<biblStruct>
<analytic>
<title level="a">Attention Is All You Need</title>
<author>
<persName>
<forename>Ashish</forename>
<surname>Vaswani</surname>
</persName>
</author>
<author>
<persName>
<forename>Noam</forename>
<surname>Shazeer</surname>
</persName>
</author>
</analytic>
<monogr>
<imprint>
<date type="published" when="2017"/>
</imprint>
</monogr>
<idno type="DOI">10.5678/attention</idno>
</biblStruct>
</listBibl>
</div>
</back>
</text>
</TEI>
"""
class _FakeResponse:
text = _TEI_XML
status_code = 200
def raise_for_status(self) -> None:
pass
def _mock_post(xml: str = _TEI_XML) -> MagicMock:
fake = MagicMock()
fake.text = xml
fake.status_code = 200
fake.raise_for_status = MagicMock()
return fake
class TestExtractReferences:
def test_returns_two_dicts(self, tmp_path: Path) -> None:
pdf = tmp_path / "paper.pdf"
pdf.write_bytes(b"%PDF fake")
with patch("httpx.Client") as mock_client_cls:
mock_client = MagicMock()
mock_client_cls.return_value.__enter__.return_value = mock_client
mock_client.post.return_value = _mock_post()
refs = extract_references(str(pdf), grobid_url="http://localhost:8070")
assert len(refs) == 2
def test_all_keys_present(self, tmp_path: Path) -> None:
pdf = tmp_path / "paper.pdf"
pdf.write_bytes(b"%PDF fake")
with patch("httpx.Client") as mock_client_cls:
mock_client = MagicMock()
mock_client_cls.return_value.__enter__.return_value = mock_client
mock_client.post.return_value = _mock_post()
refs = extract_references(str(pdf), grobid_url="http://localhost:8070")
required_keys = {"title", "authors", "year", "doi", "arxiv_id"}
for ref in refs:
assert required_keys == set(ref.keys()), f"Missing keys in {ref}"
def test_first_ref_values(self, tmp_path: Path) -> None:
pdf = tmp_path / "paper.pdf"
pdf.write_bytes(b"%PDF fake")
with patch("httpx.Client") as mock_client_cls:
mock_client = MagicMock()
mock_client_cls.return_value.__enter__.return_value = mock_client
mock_client.post.return_value = _mock_post()
refs = extract_references(str(pdf), grobid_url="http://localhost:8070")
first = refs[0]
assert first["title"] == "Deep Learning for NLP"
assert "Jane Doe" in first["authors"]
assert first["year"] == "2021"
assert first["doi"] == "10.1234/nlp.2021"
assert first["arxiv_id"] == "2101.00001"
def test_second_ref_missing_arxiv_is_empty_string(self, tmp_path: Path) -> None:
pdf = tmp_path / "paper.pdf"
pdf.write_bytes(b"%PDF fake")
with patch("httpx.Client") as mock_client_cls:
mock_client = MagicMock()
mock_client_cls.return_value.__enter__.return_value = mock_client
mock_client.post.return_value = _mock_post()
refs = extract_references(str(pdf), grobid_url="http://localhost:8070")
second = refs[1]
assert second["arxiv_id"] == ""
assert second["title"] == "Attention Is All You Need"
assert second["year"] == "2017"
class TestExtractStructure:
def test_returns_xml_string(self, tmp_path: Path) -> None:
pdf = tmp_path / "paper.pdf"
pdf.write_bytes(b"%PDF fake")
with patch("httpx.Client") as mock_client_cls:
mock_client = MagicMock()
mock_client_cls.return_value.__enter__.return_value = mock_client
mock_client.post.return_value = _mock_post("<TEI>raw xml</TEI>")
result = extract_structure(str(pdf), grobid_url="http://localhost:8070")
assert result == "<TEI>raw xml</TEI>"
mock_client.post.assert_called_once()
assert "processFulltextDocument" in mock_client.post.call_args[0][0]