Files
codex-py/tests/parsing/test_tex.py
2026-06-04 23:30:58 +02:00

113 lines
4.1 KiB
Python

"""Tests for codex.parsing.tex."""
from __future__ import annotations
import pytest
from codex.parsing.tex import chunk_text, extract_sections, latex_to_text
class TestExtractSections:
def test_two_sections_returned(self) -> None:
latex = r"\section{Intro}" + "\nHello \\cite{foo} world\n"
latex += r"\section{Method}" + "\nResult"
sections = extract_sections(latex)
assert len(sections) == 2
def test_section_titles(self) -> None:
latex = r"\section{Intro}" + "\nHello\n" + r"\section{Method}" + "\nResult"
sections = extract_sections(latex)
assert sections[0][0] == "Intro"
assert sections[1][0] == "Method"
def test_cite_removed_from_text(self) -> None:
latex = r"\section{Intro}" + "\nHello \\cite{foo} world\n"
latex += r"\section{Method}" + "\nResult"
sections = extract_sections(latex)
body = sections[0][1]
assert r"\cite" not in body
assert "foo" not in body
def test_no_sections_returns_empty(self) -> None:
assert extract_sections("No sections here at all.") == []
def test_subsection_recognised(self) -> None:
latex = r"\subsection{Background}" + "\nSome text."
sections = extract_sections(latex)
assert len(sections) == 1
assert sections[0][0] == "Background"
def test_comments_stripped(self) -> None:
latex = r"\section{A}" + "\nHello % this is a comment\nWorld"
sections = extract_sections(latex)
assert "%" not in sections[0][1]
assert "comment" not in sections[0][1]
def test_math_stripped(self) -> None:
latex = r"\section{A}" + "\nLet $x = 1$ and $$y = 2$$ be numbers."
sections = extract_sections(latex)
body = sections[0][1]
assert "$" not in body
class TestChunkText:
@pytest.fixture()
def long_text(self) -> str:
# Build a 600-word string with real sentences
sentence = "The quick brown fox jumps over the lazy dog near the river. "
words_per_sentence = len(sentence.split())
repeats = (600 // words_per_sentence) + 1
return (sentence * repeats).strip()
def test_at_least_two_chunks(self, long_text: str) -> None:
chunks = chunk_text(long_text, size=512, overlap=64)
assert len(chunks) >= 2
def test_each_chunk_within_size_bound(self, long_text: str) -> None:
size, overlap = 512, 64
chunks = chunk_text(long_text, size=size, overlap=overlap)
for chunk in chunks:
word_count = len(chunk.split())
assert word_count <= size + overlap, (
f"Chunk has {word_count} words, expected <= {size + overlap}"
)
def test_empty_text_returns_empty_list(self) -> None:
assert chunk_text("") == []
def test_short_text_returns_single_chunk(self) -> None:
chunks = chunk_text("Hello world.", size=512, overlap=64)
assert len(chunks) == 1
def test_chunks_cover_all_content(self) -> None:
# First chunk must start with first word, last chunk must end with last word
text = " ".join(f"word{i}" for i in range(600))
chunks = chunk_text(text, size=200, overlap=50)
assert chunks[0].startswith("word0")
assert chunks[-1].endswith("word599")
class TestLatexToText:
def test_no_percent_in_output(self) -> None:
latex = r"\section{A}" + "\nSome % comment\ntext."
result = latex_to_text(latex)
assert "%" not in result
def test_no_cite_in_output(self) -> None:
latex = r"\section{A}" + "\nHello \\cite{bar} world."
result = latex_to_text(latex)
assert r"\cite" not in result
assert "bar" not in result
def test_returns_string(self) -> None:
latex = r"\section{A}" + "\nSimple text."
result = latex_to_text(latex)
assert isinstance(result, str)
assert len(result) > 0
def test_no_section_fallback(self) -> None:
latex = r"Some \cite{foo} text with % comment"
result = latex_to_text(latex)
assert r"\cite" not in result
assert "%" not in result