"""
Ingest a PDF into the local Chroma collection for the Realtime voice lab.

Usage (inside .venv):
  python ingest_pdf.py
  python ingest_pdf.py --pdf ./docs/manuel-procedures-demandeurs-information.pdf --replace
"""

from __future__ import annotations

import argparse
import hashlib
import re
from pathlib import Path

from dotenv import load_dotenv
from pypdf import PdfReader

from app import get_collection

ROOT = Path(__file__).resolve().parent
DEFAULT_PDF = ROOT / "docs" / "manuel-procedures-demandeurs-information.pdf"


def clean_text(text: str) -> str:
    text = text.replace("\x00", " ")
    text = re.sub(r"[ \t]+", " ", text)
    text = re.sub(r"\n{3,}", "\n\n", text)
    return text.strip()


def chunk_text(text: str, *, chunk_size: int = 900, overlap: int = 150) -> list[str]:
    if not text:
        return []
    chunks: list[str] = []
    start = 0
    n = len(text)
    while start < n:
        end = min(start + chunk_size, n)
        # Prefer breaking on paragraph/sentence boundaries.
        if end < n:
            window = text[start:end]
            break_at = max(window.rfind("\n\n"), window.rfind(". "), window.rfind("؟"), window.rfind("."))
            if break_at > chunk_size // 3:
                end = start + break_at + 1
        piece = text[start:end].strip()
        if piece:
            chunks.append(piece)
        if end >= n:
            break
        start = max(end - overlap, start + 1)
    return chunks


def extract_pages(pdf_path: Path) -> list[tuple[int, str]]:
    reader = PdfReader(str(pdf_path))
    pages: list[tuple[int, str]] = []
    for i, page in enumerate(reader.pages, start=1):
        raw = page.extract_text() or ""
        text = clean_text(raw)
        if text:
            pages.append((i, text))
    return pages


def make_chunk_id(source: str, page: int, index: int, text: str) -> str:
    digest = hashlib.sha1(f"{source}:{page}:{index}:{text[:80]}".encode("utf-8")).hexdigest()[:12]
    return f"pdf-{digest}-p{page}-{index}"


def ingest_pdf(
    pdf_path: Path,
    *,
    replace: bool = True,
    chunk_size: int = 900,
    overlap: int = 150,
) -> dict:
    if not pdf_path.exists():
        raise FileNotFoundError(f"PDF not found: {pdf_path}")

    collection = get_collection()
    source_name = pdf_path.name

    if replace and collection.count() > 0:
        existing = collection.get()
        ids = existing.get("ids") or []
        if ids:
            # Delete in batches (Chroma can choke on huge single deletes).
            batch = 200
            for i in range(0, len(ids), batch):
                collection.delete(ids=ids[i : i + batch])

    pages = extract_pages(pdf_path)
    if not pages:
        raise RuntimeError("No extractable text found in PDF (may be scanned/image-only).")

    ids: list[str] = []
    documents: list[str] = []
    metadatas: list[dict] = []

    for page_num, page_text in pages:
        pieces = chunk_text(page_text, chunk_size=chunk_size, overlap=overlap)
        for idx, piece in enumerate(pieces):
            chunk_id = make_chunk_id(source_name, page_num, idx, piece)
            ids.append(chunk_id)
            documents.append(piece)
            metadatas.append(
                {
                    "source": source_name,
                    "title": f"{source_name} — page {page_num}",
                    "page": page_num,
                    "chunk_index": idx,
                    "doc_type": "pdf",
                }
            )

    # Upsert in batches to avoid large embedding requests.
    batch_size = 32
    for i in range(0, len(ids), batch_size):
        collection.upsert(
            ids=ids[i : i + batch_size],
            documents=documents[i : i + batch_size],
            metadatas=metadatas[i : i + batch_size],
        )

    return {
        "pdf": str(pdf_path),
        "pages_with_text": len(pages),
        "chunks": len(ids),
        "collection_count": collection.count(),
        "replaced": replace,
    }


def main() -> None:
    load_dotenv()
    parser = argparse.ArgumentParser(description="Ingest a PDF into realtime-kb Chroma")
    parser.add_argument("--pdf", type=Path, default=DEFAULT_PDF, help="Path to PDF")
    parser.add_argument(
        "--replace",
        action="store_true",
        default=True,
        help="Clear existing collection before ingest (default: true)",
    )
    parser.add_argument(
        "--keep-existing",
        action="store_true",
        help="Do not clear existing documents",
    )
    parser.add_argument("--chunk-size", type=int, default=900)
    parser.add_argument("--overlap", type=int, default=150)
    args = parser.parse_args()

    replace = False if args.keep_existing else args.replace
    result = ingest_pdf(
        args.pdf,
        replace=replace,
        chunk_size=args.chunk_size,
        overlap=args.overlap,
    )
    print(result)


if __name__ == "__main__":
    main()
