"""Domain dataclasses. Each class maps directly to a table in ``infra/schema.sql``. Field names and types are kept in sync with the DDL. All fields that are nullable in the schema are typed as ``… | None`` with a default of ``None`` so instances can be constructed incrementally. """ from __future__ import annotations from dataclasses import dataclass, field from datetime import datetime @dataclass class Paper: """Maps to the ``papers`` table (Layer 1 — semantic search). ``id`` is the canonical identifier: an arXiv ID (e.g. ``2301.00001``) or a DOI (e.g. ``10.1145/3592430``). """ id: str title: str openalex_id: str | None = None bibkey: str | None = None authors: list[str] = field(default_factory=list) year: int | None = None abstract: str | None = None source_path: str | None = None # abstract_emb is a list of floats representing the pgvector column; # pgvector's Python driver accepts list[float] for vector columns. abstract_emb: list[float] | None = None added_at: datetime | None = None @dataclass class Chunk: """Maps to the ``chunks`` table (Layer 1 — full-text chunks). ``id`` is set by the database (BIGSERIAL); leave it as ``None`` when constructing a new chunk before insertion. """ paper_id: str ord: int content: str id: int | None = None embedding: list[float] | None = None @dataclass class Citation: """Maps to the ``citations`` table (Layer 2 — citation graph). ``cited_id`` intentionally has no foreign-key constraint in the schema; citations to not-yet-ingested papers are preserved as discovery leads. """ citing_id: str cited_id: str context: str | None = None @dataclass class CodeLink: """Maps to the ``code_links`` table (Layer 3 — provenance). ``symbol`` is a C++ qualified name or ``file.cpp:line`` reference. ``role`` is a free-text description such as 'implements Thm 3.2'. ``id`` is set by the database (BIGSERIAL). """ symbol: str paper_id: str | None = None role: str | None = None note: str | None = None id: int | None = None added_at: datetime | None = None