#!/usr/bin/env python3
"""Ingest PDF documents into ChromaDB for RAG retrieval.

Usage:
    python scripts/ingest_pdf.py path/to/file.pdf
    python scripts/ingest_pdf.py docs/          # ingest all PDFs in a folder
    python scripts/ingest_pdf.py docs/ --reset  # clear existing PDF docs first
"""

import argparse
import hashlib
import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent.parent))

try:
    import pdfplumber  # type: ignore
except ImportError:
    print("pdfplumber not installed. Run: pip install pdfplumber")
    sys.exit(1)

from langchain_core.documents import Document
from tqdm import tqdm

from app.config import get_settings
from app.rag.chain import RAGService

CHUNK_SIZE = 1200   # characters per chunk
CHUNK_OVERLAP = 200


def extract_pages(pdf_path: Path) -> list[str]:
    """Extract text page-by-page from a PDF."""
    pages: list[str] = []
    with pdfplumber.open(str(pdf_path)) as pdf:
        for page in pdf.pages:
            text = page.extract_text() or ""
            text = text.strip()
            if text:
                pages.append(text)
    return pages


def chunk_text(text: str, chunk_size: int = CHUNK_SIZE, overlap: int = CHUNK_OVERLAP) -> list[str]:
    """Split text into overlapping character chunks."""
    chunks: list[str] = []
    start = 0
    while start < len(text):
        end = min(start + chunk_size, len(text))
        chunks.append(text[start:end])
        if end == len(text):
            break
        start += chunk_size - overlap
    return chunks


def pdf_to_documents(pdf_path: Path) -> list[Document]:
    pages = extract_pages(pdf_path)
    docs: list[Document] = []
    for page_num, page_text in enumerate(pages, start=1):
        for chunk_idx, chunk in enumerate(chunk_text(page_text)):
            # Stable record_id based on file + page + chunk position
            chunk_hash = hashlib.md5(f"{pdf_path.name}:{page_num}:{chunk_idx}".encode()).hexdigest()[:12]
            record_id = f"pdf_{pdf_path.stem}_p{page_num}_c{chunk_idx}_{chunk_hash}"
            docs.append(
                Document(
                    page_content=chunk,
                    metadata={
                        "record_id": record_id,
                        "data_category": "policy_document",
                        "source_dataset": pdf_path.name,
                        "page": page_num,
                    },
                )
            )
    return docs


def ingest_pdf(pdf_path: Path, service: RAGService, already_indexed: set[str]) -> int:
    docs = pdf_to_documents(pdf_path)
    new_docs = [d for d in docs if d.metadata["record_id"] not in already_indexed]
    if not new_docs:
        print(f"  {pdf_path.name}: already fully indexed ({len(docs)} chunks), skipping.")
        return 0
    for doc in tqdm(new_docs, desc=f"  {pdf_path.name}", unit="chunk"):
        service.get_vectorstore().add_documents([doc])
    return len(new_docs)


def get_indexed_record_ids(service: RAGService) -> set[str]:
    store = service.get_vectorstore()
    collection = store._collection
    indexed: set[str] = set()
    offset = 0
    page_size = 5000
    while True:
        batch = collection.get(include=["metadatas"], limit=page_size, offset=offset)
        metadatas = batch.get("metadatas") or []
        if not metadatas:
            break
        for meta in metadatas:
            if meta and meta.get("record_id"):
                indexed.add(meta["record_id"])
        if len(metadatas) < page_size:
            break
        offset += page_size
    return indexed


def main() -> None:
    parser = argparse.ArgumentParser(description="Ingest PDF documents into ChromaDB")
    parser.add_argument("path", help="Path to a PDF file or directory of PDFs")
    parser.add_argument("--reset", action="store_true", help="Clear existing PDF documents before ingesting")
    args = parser.parse_args()

    target = Path(args.path)
    if target.is_dir():
        pdf_files = sorted(target.glob("**/*.pdf"))
    elif target.suffix.lower() == ".pdf":
        pdf_files = [target]
    else:
        print(f"Error: {target} is not a PDF file or directory")
        sys.exit(1)

    if not pdf_files:
        print("No PDF files found.")
        sys.exit(0)

    settings = get_settings()
    service = RAGService(settings)

    if args.reset:
        # Remove only PDF-sourced documents
        store = service.get_vectorstore()
        collection = store._collection
        # Get IDs of PDF docs
        batch = collection.get(include=["metadatas", "ids"])
        ids_to_delete = [
            id_ for id_, meta in zip(batch.get("ids", []), batch.get("metadatas", []))
            if meta and meta.get("data_category") == "policy_document"
        ]
        if ids_to_delete:
            collection.delete(ids=ids_to_delete)
            print(f"Deleted {len(ids_to_delete)} existing PDF chunks.")

    already_indexed = get_indexed_record_ids(service)
    total_new = 0

    print(f"Found {len(pdf_files)} PDF file(s).")
    for pdf_path in pdf_files:
        print(f"Processing {pdf_path.name}...")
        try:
            added = ingest_pdf(pdf_path, service, already_indexed)
            total_new += added
        except Exception as e:
            print(f"  ERROR processing {pdf_path.name}: {e}")

    final_count = service.get_vectorstore()._collection.count()
    print(f"\nDone. Added {total_new:,} new chunks.")
    print(f"ChromaDB collection now has {final_count:,} documents total.")


if __name__ == "__main__":
    main()
