diff --git a/codex/wiki.py b/codex/wiki.py index 65f988b..8a386e5 100644 --- a/codex/wiki.py +++ b/codex/wiki.py @@ -228,6 +228,10 @@ def _parse_claims(markdown: str) -> list[Claim]: text = match.group("text").strip() bibkey = match.group("bibkey").strip() locator = (match.group("locator") or "").strip() + # Skip URL-shaped bibkeys ([text](https://...)) and multi-word bibkeys + # (real BibKeys never contain spaces or start with "http") + if " " in bibkey or bibkey.startswith("http"): + continue if text and bibkey: claims.append(Claim(text=text, bibkey=bibkey, locator=locator)) return claims diff --git a/tests/wiki/test_compile.py b/tests/wiki/test_compile.py index e2f5ad5..89498be 100644 --- a/tests/wiki/test_compile.py +++ b/tests/wiki/test_compile.py @@ -121,6 +121,20 @@ def test_parse_claims_multiple() -> None: assert len(claims) == 2 +def test_parse_claims_ignores_markdown_links() -> None: + """Markdown hyperlinks like [text](https://example.com) are NOT parsed as claims.""" + md = "See the [related work](https://example.com) for details." + claims = _parse_claims(md) + assert claims == [] + + +def test_parse_claims_ignores_multi_word_bibkey() -> None: + """Multi-word 'bibkeys' (spaces inside) are not treated as real citations.""" + md = "Some text [not a bibkey here] and more." + claims = _parse_claims(md) + assert claims == [] + + # --------------------------------------------------------------------------- # _run_grounding_guard # ---------------------------------------------------------------------------