feat(parsing): LaTeX, Nougat, GROBID parsers with chunking

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
Tarik Moussa
2026-06-04 23:30:58 +02:00
parent b52a6a7412
commit c6a428d335
8 changed files with 672 additions and 1 deletions

View File

@@ -9,7 +9,7 @@ from __future__ import annotations
from functools import lru_cache
from pydantic import Field
from pydantic import AliasChoices, Field
from pydantic_settings import BaseSettings, SettingsConfigDict
@@ -42,6 +42,12 @@ class Settings(BaseSettings):
description="Base URL of the GROBID HTTP API (containerised).",
)
nougat_url: str = Field(
default="http://localhost:8080",
validation_alias=AliasChoices("NOUGAT_URL", "nougat_url"),
description="Base URL of the Nougat OCR HTTP API (containerised).",
)
ollama_base_url: str = Field(
default="http://localhost:11434",
description="Base URL of the local Ollama endpoint (optional Q&A layer).",

131
codex/parsing/grobid.py Normal file
View File

@@ -0,0 +1,131 @@
"""GROBID integration.
Extracts structured reference lists and full-text TEI XML from PDFs by
calling a self-hosted GROBID HTTP server. No DB access, no embedding,
no network fetching of papers happens here.
"""
from __future__ import annotations
import xml.etree.ElementTree as ET
import httpx
from codex.config import Settings
_TEI_NS = "{http://www.tei-c.org/ns/1.0}"
def _text(element: ET.Element | None) -> str:
"""Return element text or empty string if element is None."""
if element is None:
return ""
return (element.text or "").strip()
def extract_references(
pdf_path: str,
grobid_url: str | None = None,
) -> list[dict[str, str]]:
"""Extract a structured reference list from a PDF via GROBID.
Parameters
----------
pdf_path:
Path to the PDF file on disk.
grobid_url:
Base URL of the GROBID server. Defaults to ``Settings().grobid_url``.
Returns
-------
list[dict[str, str]]
One dict per reference. All dicts contain the keys
``title``, ``authors``, ``year``, ``doi``, ``arxiv_id``
(missing values are empty strings).
"""
if grobid_url is None:
grobid_url = Settings().grobid_url
with open(pdf_path, "rb") as fh, httpx.Client(timeout=60.0) as client:
response = client.post(
f"{grobid_url}/api/processReferences",
files={"input": (pdf_path, fh, "application/pdf")},
)
response.raise_for_status()
root = ET.fromstring(response.text)
results: list[dict[str, str]] = []
for bib in root.iter(f"{_TEI_NS}biblStruct"):
# Title
title_el = bib.find(f".//{_TEI_NS}title[@level='a']")
title = _text(title_el)
# Authors: collect all persName elements
authors_parts: list[str] = []
for person in bib.iter(f"{_TEI_NS}persName"):
forename_el = person.find(f"{_TEI_NS}forename")
surname_el = person.find(f"{_TEI_NS}surname")
forename = _text(forename_el)
surname = _text(surname_el)
full = " ".join(p for p in (forename, surname) if p)
if full:
authors_parts.append(full)
authors = "; ".join(authors_parts)
# Year
date_el = bib.find(f".//{_TEI_NS}date[@type='published']")
year = ""
if date_el is not None:
when = date_el.get("when", "")
year = when[:4] if when else ""
# DOI
doi_el = bib.find(f".//{_TEI_NS}idno[@type='DOI']")
doi = _text(doi_el)
# arXiv ID
arxiv_el = bib.find(f".//{_TEI_NS}idno[@type='arxiv']")
arxiv_id = _text(arxiv_el)
results.append(
{
"title": title,
"authors": authors,
"year": year,
"doi": doi,
"arxiv_id": arxiv_id,
}
)
return results
def extract_structure(
pdf_path: str,
grobid_url: str | None = None,
) -> str:
"""Extract full-text TEI XML from a PDF via GROBID.
Parameters
----------
pdf_path:
Path to the PDF file on disk.
grobid_url:
Base URL of the GROBID server. Defaults to ``Settings().grobid_url``.
Returns
-------
str
Raw TEI XML response text from GROBID.
"""
if grobid_url is None:
grobid_url = Settings().grobid_url
with open(pdf_path, "rb") as fh, httpx.Client(timeout=120.0) as client:
response = client.post(
f"{grobid_url}/api/processFulltextDocument",
files={"input": (pdf_path, fh, "application/pdf")},
)
response.raise_for_status()
return response.text

54
codex/parsing/nougat.py Normal file
View File

@@ -0,0 +1,54 @@
"""Nougat OCR integration.
Converts PDF files to Mathpix Markdown (mmd) by calling a self-hosted
Nougat HTTP server. No network fetching of papers happens here — the
caller is expected to pass a local PDF path.
"""
from __future__ import annotations
import httpx
import tenacity
from codex.config import Settings
def pdf_to_markdown(pdf_path: str, nougat_url: str | None = None) -> str:
"""Convert a local PDF to Mathpix Markdown via the Nougat HTTP API.
Parameters
----------
pdf_path:
Absolute (or relative) path to the PDF file on disk.
nougat_url:
Base URL of the Nougat server. Defaults to ``Settings().nougat_url``.
Returns
-------
str
The raw ``.mmd`` text returned by the server.
Raises
------
httpx.HTTPStatusError
If the server returns a non-2xx status code.
"""
if nougat_url is None:
nougat_url = Settings().nougat_url
@tenacity.retry(
retry=tenacity.retry_if_exception_type(httpx.ConnectError),
stop=tenacity.stop_after_attempt(3), # 1 original + 2 retries
wait=tenacity.wait_fixed(0),
reraise=True,
)
def _post() -> str:
with open(pdf_path, "rb") as fh, httpx.Client(timeout=120.0) as client:
response = client.post(
f"{nougat_url}/predict",
files={"file": (pdf_path, fh, "application/pdf")},
)
response.raise_for_status()
return response.text
return _post()

159
codex/parsing/tex.py Normal file
View File

@@ -0,0 +1,159 @@
"""LaTeX parsing utilities.
Provides helpers to extract sections, chunk text, and convert LaTeX to plain
readable prose by stripping markup that is not useful for NLP/embedding.
"""
from __future__ import annotations
import re
# ---------------------------------------------------------------------------
# Internal helpers
# ---------------------------------------------------------------------------
_SECTION_RE = re.compile(
r"\\(?:sub)*section\*?\s*\{([^}]*)\}",
re.DOTALL,
)
# Patterns for LaTeX noise removal (applied in order).
_COMMENT_RE = re.compile(r"%[^\n]*")
_CITE_RE = re.compile(r"\\cite\*?\{[^}]*\}")
_LABEL_RE = re.compile(r"\\label\{[^}]*\}")
_REF_RE = re.compile(r"\\ref\{[^}]*\}")
# Display math: $$...$$ (before single $ to avoid greedy mismatch)
_DISPLAY_DOLLAR_RE = re.compile(r"\$\$.*?\$\$", re.DOTALL)
# Inline math: $...$
_INLINE_MATH_RE = re.compile(r"\$[^$\n]*?\$", re.DOTALL)
# \begin{equation}...\end{equation}
_ENV_EQUATION_RE = re.compile(
r"\\begin\{equation\*?\}.*?\\end\{equation\*?\}",
re.DOTALL,
)
# \begin{align}...\end{align} (covers align, align*, aligned, …)
_ENV_ALIGN_RE = re.compile(
r"\\begin\{align[^}]*\}.*?\\end\{align[^}]*\}",
re.DOTALL,
)
# Collapse excess whitespace
_WHITESPACE_RE = re.compile(r"[ \t]+")
_BLANK_LINES_RE = re.compile(r"\n{3,}")
def _clean_latex(text: str) -> str:
"""Strip LaTeX markup and return readable prose."""
text = _COMMENT_RE.sub("", text)
text = _ENV_EQUATION_RE.sub("", text)
text = _ENV_ALIGN_RE.sub("", text)
text = _DISPLAY_DOLLAR_RE.sub("", text)
text = _INLINE_MATH_RE.sub("", text)
text = _CITE_RE.sub("", text)
text = _LABEL_RE.sub("", text)
text = _REF_RE.sub("", text)
text = _WHITESPACE_RE.sub(" ", text)
text = _BLANK_LINES_RE.sub("\n\n", text)
return text.strip()
# ---------------------------------------------------------------------------
# Public API
# ---------------------------------------------------------------------------
def extract_sections(latex: str) -> list[tuple[str, str]]:
"""Split *latex* on \\section / \\subsection boundaries.
Returns a list of ``(title, cleaned_text)`` tuples, one per section.
Text between the document start and the first section command is
discarded (preamble / abstract handling is out of scope).
"""
# Find all section positions
matches = list(_SECTION_RE.finditer(latex))
if not matches:
return []
sections: list[tuple[str, str]] = []
for idx, match in enumerate(matches):
title = match.group(1).strip()
body_start = match.end()
body_end = matches[idx + 1].start() if idx + 1 < len(matches) else len(latex)
body = latex[body_start:body_end]
cleaned = _clean_latex(body)
sections.append((title, cleaned))
return sections
def chunk_text(text: str, size: int = 512, overlap: int = 64) -> list[str]:
"""Split *text* into overlapping word-based chunks.
Parameters
----------
text:
Plain text to chunk (not raw LaTeX).
size:
Target number of words per chunk.
overlap:
Number of words to carry over into the next chunk.
Each chunk is at most ``size + overlap`` words. Boundaries are snapped
to the nearest ``". "`` (sentence end) within ±20 words of the nominal
boundary when possible.
"""
words = text.split()
if not words:
return []
snap_window = 20
chunks: list[str] = []
start = 0
while start < len(words):
end = min(start + size, len(words))
# Try to snap *end* to a sentence boundary within ±snap_window words.
if end < len(words):
best = end
# Build a small search window
lo = max(start + 1, end - snap_window)
hi = min(len(words), end + snap_window + 1)
# Prefer the closest sentence-end ". " to *end*
for offset in range(0, snap_window + 1):
for candidate in (end - offset, end + offset):
if (
lo <= candidate < hi
and candidate > start
and words[candidate - 1].endswith(".")
):
# Sentence boundary: word at (candidate-1) ends with ".".
best = candidate
break
if best != end:
break
end = best
chunk_words = words[start:end]
chunks.append(" ".join(chunk_words))
if end >= len(words):
break
# Next chunk starts *overlap* words before *end*.
start = max(start + 1, end - overlap)
return chunks
def latex_to_text(latex: str) -> str:
"""Convert a LaTeX document to plain text.
Extracts all sections, cleans each one, and joins them with a blank line.
If no section commands are found, the whole document is cleaned and
returned as a single block.
"""
sections = extract_sections(latex)
if sections:
return "\n\n".join(body for _, body in sections)
# Fallback: no section structure — clean the whole string.
return _clean_latex(latex)