"""Knowledge-awareness for SaaS chat — empty KB vs no match vs found."""

from __future__ import annotations

import re
import time
from dataclasses import dataclass
from typing import Literal

from app.saas import db as saas_db

KnowledgeState = Literal["empty", "no_relevant", "found"]

EMPTY_KB_MESSAGE = (
    "I don't have a knowledge base yet, so I can't answer questions about your finances "
    "or documents.\n\n"
    "Please upload a knowledge base first:\n"
    "• Upload a PDF statement\n"
    "• Upload a CSV export\n"
    "• Add files in the Knowledge panel"
)

NO_RELEVANT_MESSAGE = (
    "I couldn't find any information related to your question in your uploaded knowledge base.\n\n"
    "Try rephrasing, ask about totals or a merchant on your statement, or upload another document."
)

_KB_META_RE = re.compile(
    r"(?i)\b("
    r"knowledge\s*base|my\s+(?:documents?|files?|uploads?|pdfs?)|"
    r"what\s+(?:documents?|files?)\s+(?:do\s+you\s+have|are\s+uploaded|are\s+ready|are\s+indexed)|"
    r"uploaded\s+(?:documents?|files?|pdfs?)|"
    r"which\s+(?:statement|pdf|document|file|upload)|"
    r"what\s+(?:statement|pdf|document|file)\s+(?:is\s+)?(?:this|that|loaded|ready|uploaded)|"
    r"what\s+(?:statement|pdf|document)\s+do\s+(?:you|i)\s+have|"
    r"what\s+file\s+is\s+(?:this|loaded)|"
    r"how\s+many\s+(?:documents?|files?|pdfs?|pages?|chunks?)|"
    r"(?:page|pages)\s+count|number\s+of\s+(?:pages?|documents?|files?)|"
    r"what(?:'s|\s+is)\s+the\s+filename|(?:file|document|pdf)\s+name|"
    r"upload\s+status|indexing\s+status|"
    r"is\s+(?:it|the\s+(?:file|pdf|document))\s+(?:ready|uploaded|indexed)|"
    r"list\s+(?:my\s+)?(?:documents?|files?|uploads?)|"
    r"tell\s+me\s+about\s+(?:the\s+)?knowledge\s*base"
    r")\b"
)

# Short TTL cache — avoids remote DB + Chroma on every chat turn
_STATUS_CACHE: dict[str, tuple[float, "KnowledgeStatus"]] = {}
_STATUS_TTL_S = 45.0


@dataclass(frozen=True)
class KnowledgeStatus:
    """Whether the selected agent has any indexed / structured knowledge."""

    vector_count: int
    ready_docs: int
    ready_chunks: int
    statement_jobs: int

    @property
    def has_knowledge(self) -> bool:
        return (
            self.vector_count > 0
            or self.ready_docs > 0
            or self.ready_chunks > 0
            or self.statement_jobs > 0
        )


def unique_ready_docs(docs: list[dict]) -> list[dict]:
    """Collapse duplicate ready uploads of the same filename (keep newest)."""
    best: dict[str, dict] = {}
    for d in docs or []:
        if d.get("status") != "ready":
            continue
        key = re.sub(r"\s+", " ", str(d.get("filename") or "").strip().lower())
        if not key:
            continue
        stamp = str(d.get("updated_at") or d.get("created_at") or "")
        prev = best.get(key)
        if not prev or stamp >= str(prev.get("updated_at") or prev.get("created_at") or ""):
            best[key] = d
    return list(best.values())


def invalidate_knowledge_cache(org_id: str | None = None, agent_id: str | None = None) -> None:
    """Drop cached KB status after upload / delete / ingest."""
    if org_id and agent_id:
        _STATUS_CACHE.pop(f"{org_id}:{agent_id}", None)
        return
    if org_id:
        prefix = f"{org_id}:"
        for key in list(_STATUS_CACHE):
            if key.startswith(prefix):
                _STATUS_CACHE.pop(key, None)
        return
    _STATUS_CACHE.clear()


def assess_agent_knowledge(org_id: str, agent_id: str) -> KnowledgeStatus:
    """Inspect the agent's KB before retrieval or LLM generation.

    Uses a single documents query (no Chroma open, no PDF re-extract) so chat
    stays fast on remote Postgres.
    """
    key = f"{org_id}:{agent_id}"
    hit = _STATUS_CACHE.get(key)
    now = time.monotonic()
    if hit and (now - hit[0]) < _STATUS_TTL_S:
        return hit[1]

    try:
        docs = saas_db.list_kb_documents(org_id, agent_id) or []
    except Exception:
        docs = []

    ready = [d for d in docs if d.get("status") == "ready"]
    unique = unique_ready_docs(ready)
    ready_docs = len(unique)
    ready_chunks = sum(int(d.get("chunk_count") or 0) for d in unique)
    # PDF ready docs imply statement analysis is available (extract is cached later)
    statement_jobs = sum(
        1 for d in unique if str(d.get("filename") or "").lower().endswith(".pdf")
    )
    # Proxy for vectors without opening Chroma on every turn
    vector_count = ready_chunks if ready_chunks > 0 else (1 if ready_docs else 0)

    status = KnowledgeStatus(
        vector_count=vector_count,
        ready_docs=ready_docs,
        ready_chunks=ready_chunks,
        statement_jobs=statement_jobs,
    )
    _STATUS_CACHE[key] = (now, status)
    return status


def is_kb_meta_question(question: str) -> bool:
    """True when the user is asking about the agent's uploaded knowledge itself."""
    q = question or ""
    # “Based on the knowledge base, what is my name?” is content — not inventory.
    if re.search(
        r"(?i)\b(?:my\s+name|account\s+holder|who\s+am\s+i|whose\s+statement|"
        r"how\s+much|total\s+spent|biggest\s+expenses?)\b",
        q,
    ):
        return False
    return bool(_KB_META_RE.search(q))


def needs_knowledge(finance_intent: str, chat_intent: str, question: str = "") -> bool:
    """True when the question requires the agent's uploaded knowledge base.

    Document metadata and document content need the KB.
    Conversational / general knowledge / pure math do not.
    """
    from app.saas.intent_router import classify_primary_intent

    q = (question or "").strip()
    # Legacy callers sometimes omit the question and only pass a finance intent.
    if not q and finance_intent and finance_intent not in {"general"}:
        if chat_intent in {"greeting", "thanks", "bye", "math", "concept", "offtopic", "help"}:
            return False
        return True
    if not q:
        return False
    primary = classify_primary_intent(q, finance_intent=finance_intent or None)
    return primary.intent in {"document_lookup", "document_metadata"}
