"""Primary intent classification — decide the channel BEFORE any retrieval.

Five intents (exactly one per turn):
  1. document_lookup   — answer depends on uploaded PDF/statement contents → structured KB / Chroma
  2. document_metadata — filenames, page counts, doc count, upload/index status → inventory only
  3. general_knowledge — concepts, geography, science, finance theory → base LLM (no RAG)
  4. calculation       — pure math or hypothetical numeric scenarios → calculator / LLM math
  5. conversational    — greetings, smalltalk, capability, personal memory → templates

Only (1) and (2) touch the knowledge base. (2) never summarizes document contents.
"""

from __future__ import annotations

import re
from dataclasses import dataclass
from typing import Literal

from app.pdf.calc import is_pdf_overview_question
from app.rag.calculator import is_pure_math
from app.rag.chain import (
    _BYE_WORDS,
    _GREETING_WORDS,
    _HELP_WORDS,
    _HOW_ARE_YOU_RE,
    _OFFTOPIC_RE,
    _THANKS_WORDS,
    _detect_intent,
    _likely_greeting,
    _normalize_smalltalk,
)
from app.saas.finance_intent import (
    classify_finance_intent,
    is_analyse_overview_ask,
    is_analytical_self_contained,
    is_biggest_expenses_question,
    is_payment_list_question,
    is_period_coverage_question,
    is_profit_loss_question,
    is_statement_field_question,
    is_statement_identity_question,
    is_statement_period_question,
    is_unrelated_party_spend_question,
)
from app.saas.knowledge_state import is_kb_meta_question

PrimaryIntent = Literal[
    "document_lookup",
    "document_metadata",
    "general_knowledge",
    "calculation",
    "conversational",
]

_VAGUE_PING_RE = re.compile(
    r"(?i)^(what|huh|eh|and|so|ok|okay|um+|hmm*|complete|done|yes|yep|yeah|"
    r"sure|cool|nice|great|alright|right|got\s+it|fine|k|kk|tell\s+me|"
    r"[?!.…]+|what\?+|huh\?+)\s*$"
)

_SELF_INTRO_RE = re.compile(
    r"(?i)^\s*(?:hi[,!]?\s+|hello[,!]?\s+)?"
    r"(?:my\s+name\s+is|i(?:'m|\s+am|\s*m)\s+(?:called\s+)?)\s*"
    r"([A-Za-z][A-Za-z'\-]{1,30})"
    r"(?:\s+[A-Za-z][A-Za-z'\-]{1,30})?"
    r"\s*[!.?]*\s*$"
)
_ASK_MY_NAME_RE = re.compile(
    r"(?i)^\s*(?:"
    r"what(?:'s|\s+is)\s+my\s+name|"
    r"what\s+do\s+you\s+think\s+(?:is\s+)?my\s+name|"
    r"what\s+do\s+you\s+think\s+my\s+name\s+is|"
    r"(?:can\s+you\s+)?(?:guess|infer)\s+my\s+name|"
    r"who\s+am\s+i|"
    r"do\s+you\s+know\s+my\s+name|"
    r"remind\s+me\s+(?:of\s+)?my\s+name"
    r")\s*[?.!]?\s*$"
)
_ASK_FIRST_QUESTION_RE = re.compile(
    r"(?i)^\s*(?:what\s+was\s+my\s+first\s+question|"
    r"what\s+did\s+i\s+(?:ask|say)\s+first|"
    r"remind\s+me\s+(?:of\s+)?my\s+first\s+question)\s*[?.!]?\s*$"
)
_TODAY_DATE_RE = re.compile(
    r"(?i)^\s*what(?:'s|\s+is)\s+(?:the\s+)?date\s+"
    r"(?:today|todat|todqy|now|currently)\s*[?.!]?\s*$"
    r"|^\s*what(?:'s|\s+is)\s+today'?s\s+date\s*[?.!]?\s*$"
    r"|^\s*(?:today'?s\s+date|date\s+today)\s*[?.!]*\s*$"
)

# Document metadata — inventory / status, not content.
_DOC_METADATA_RE = re.compile(
    r"(?i)\b("
    r"knowledge\s*base|"
    r"how\s+many\s+(?:documents?|files?|pdfs?|pages?|chunks?)|"
    r"(?:page|pages)\s+count|number\s+of\s+(?:pages?|documents?|files?)|"
    r"what(?:'s|\s+is)\s+the\s+filename|"
    r"(?:file|document|pdf)\s+name|"
    r"which\s+(?:files?|documents?|pdfs?)\s+(?:are|do|did)|"
    r"what\s+(?:documents?|files?|pdfs?)\s+(?:do\s+you\s+have|are\s+(?:uploaded|ready|indexed|loaded))|"
    r"upload\s+status|indexing\s+status|is\s+(?:it|the\s+(?:file|pdf|document))\s+(?:ready|uploaded|indexed)|"
    r"my\s+(?:documents?|files?|uploads?|pdfs?)\b|"
    r"uploaded\s+(?:documents?|files?|pdfs?)|"
    r"which\s+(?:statement|pdf|document|file|upload)\b|"
    r"what\s+(?:statement|pdf|document|file)\s+(?:is\s+)?(?:this|that|loaded|ready|uploaded)|"
    r"what\s+(?:statement|pdf|document)\s+do\s+(?:you|i)\s+have|"
    r"what\s+file\s+is\s+(?:this|loaded)|"
    r"tell\s+me\s+about\s+(?:the\s+)?knowledge\s*base|"
    r"list\s+(?:my\s+)?(?:documents?|files?|uploads?)"
    r")\b"
)

# Personal / statement content that needs the uploaded PDF.
_DOC_LOOKUP_RE = re.compile(
    r"(?i)\b("
    r"my\s+(?:spend|spending|statement|payment|payments|account|balance|transaction|"
    r"transactions|invoice|outflow|inflow|merchant)|"
    r"i\s+(?:spent|spend|paid|pay|received|receive|owe)|"
    r"how\s+much\s+(?:did\s+i|have\s+i|was\s+spent|spent)|"
    r"how\s+many\s+(?:payments?|transactions?|times|credits?|deposits?|debits?)|"
    r"this\s+(?:statement|pdf|document|file)|"
    r"passbook|upi\s+statement|"
    r"spent\s+on|paid\s+to|sent\s+to|money\s+(?:in|out)|"
    r"branch\s*(?:code|number|no\.?)?|account\s*(?:number|no\.?|branch)|"
    r"ifsc|micr|cust(?:omer)?\s*id|"
    r"zepto|blinkit|swiggy|zomato|myntra|nykaa|instamart|meesho|"
    r"grocer(?:y|ies)|merchant|category\s+spend|"
    r"biggest\s+expenses?|largest\s+outflows?|"
    r"\d{1,2}(?:st|nd|rd|th)?\s+"
    r"(?:jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*|"
    r"(?:jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\s+\d{1,2}"
    r")\b"
)

_HYPOTHETICAL_RE = re.compile(
    r"(?i)\b("
    r"what\s+if|suppose\s+(?:i|we)|hypothetic|"
    r"if\s+i\s+(?:cut|reduce|increase|spent|spend|saved|save)|"
    r"(?:would|could)\s+(?:my|the)\s+(?:net|cashflow|balance|spend)|"
    r"\d+\s*%\s+(?:less|more|of)|percent\s+(?:less|more)"
    r")\b"
)

_GENERAL_KNOWLEDGE_RE = re.compile(
    r"(?i)\b("
    r"capital\s+of|who\s+invented|who\s+discovered|photosynthesis|gravity|"
    r"solar\s+system|planet\b|continent|ocean|mountain|president\s+of|"
    r"population\s+of|distance\s+between|fun\s+fact|tell\s+me\s+a\s+joke|"
    r"weather|football|cricket|movie|song|joke|recipe|horoscope|"
    r"ebitda|gaap|ifrs|gst|vat|depreciation|amorti[sz]ation|accrual|"
    r"working\s+capital|current\s+ratio|debt[\s-]?to[\s-]?equity|roi|roe|npv|irr"
    r")\b"
)

_CONCEPT_LEAD_RE = re.compile(
    r"(?i)^\s*("
    r"what\s+is|what\s+are|what'?s|define|definition\s+of|explain|meaning\s+of|"
    r"how\s+to|how\s+does|how\s+do(?:\s+i|\s+you)?|how\s+can\s+(?:i|you|one|we)\b|"
    r"difference\s+between|why\s+is|why\s+are|"
    r"who\s+(?:is|was|invented|discovered|wrote|painted)|"
    r"where\s+is|when\s+was|when\s+did|"
    r"tell\s+me\s+about(?!\s+(?:my|the\s+statement|the\s+pdf|the\s+document|this\s+pdf|the\s+knowledge))"
    r")\b"
)

# Everyday how-to / science / nature — never statement RAG.
_HOWTO_GENERAL_RE = re.compile(
    r"(?i)^\s*how\s+(?:to|do\s+i|can\s+i|does\s+one)\s+"
    r"(?!much\b|many\b|often\b|"
    r"(?:i\s+)?(?:spend|save|pay|reduce|cut|budget)|"
    r"(?:did|do)\s+i\s)"
)

# Search / KB terminology — not stock-market "index".
_KB_TECH_CONCEPT_RE = re.compile(
    r"(?i)\b("
    r"indexing|embeddings?|vector\s*(?:search|store|database)?|semantic\s+search|"
    r"chunking|retrieval|rag\b|knowledge\s+base\s+(?:index|search|work)"
    r")\b"
)

_FINANCE_INDEX_RE = re.compile(
    r"(?i)\b(stock|market|fund|portfolio|s\s*&\s*p|nasdaq|benchmark\s+index)\b"
)


@dataclass(frozen=True)
class PrimaryRoute:
    intent: PrimaryIntent
    reason: str
    finance_intent: str
    chat_intent: str


def is_self_intro(question: str) -> str | None:
    m = _SELF_INTRO_RE.match((question or "").strip())
    if not m:
        return None
    name = m.group(1).strip()
    if name.lower() in {"alex", "you", "here", "good", "fine", "ok", "okay"}:
        return None
    return name.title()


def is_kb_tech_concept_question(question: str) -> bool:
    """Document/search indexing concepts — not financial market indexes."""
    q = (question or "").strip()
    if not q or not _CONCEPT_LEAD_RE.search(q):
        return False
    if not _KB_TECH_CONCEPT_RE.search(q):
        return False
    return not _FINANCE_INDEX_RE.search(q)


def is_ask_my_name(question: str) -> bool:
    return bool(_ASK_MY_NAME_RE.match((question or "").strip()))


def is_ask_first_question(question: str) -> bool:
    return bool(_ASK_FIRST_QUESTION_RE.match((question or "").strip()))


def is_today_date_question(question: str) -> bool:
    return bool(_TODAY_DATE_RE.match((question or "").strip()))


def is_document_metadata_question(question: str) -> bool:
    """Inventory / status of indexed docs — not a content summary."""
    q = question or ""
    # Content asks that mention “knowledge base” as the source — not inventory.
    if is_statement_identity_question(q) or is_statement_period_question(q) or is_statement_field_question(q):
        return False
    if re.search(
        r"(?i)\b(?:my\s+name|account\s+holder|who\s+am\s+i|whose\s+statement|"
        r"how\s+much|total\s+spent|biggest\s+expenses?)\b",
        q,
    ):
        return False
    # KB inventory always wins over generic “tell me about …” overview detectors.
    if is_kb_meta_question(q) or _DOC_METADATA_RE.search(q):
        if re.search(
            r"(?i)\b(analy[sz]e|summar(?:y|ize)|overview)\b.*\b(statement|pdf|document|spend)\b",
            q,
        ):
            return False
        if re.search(r"(?i)\b(spent|spend|paid|how\s+much)\b", q) and not re.search(
            r"(?i)\b(how\s+many\s+(?:pages?|documents?|files?)|page\s+count|filename|upload\s+status|knowledge\s*base)\b",
            q,
        ):
            return False
        return True
    return False


def is_hypothetical_or_calc(question: str) -> bool:
    q = (question or "").strip()
    if is_pure_math(q):
        return True
    if _HYPOTHETICAL_RE.search(q) and not _DOC_LOOKUP_RE.search(q):
        return True
    # "what is 15% of 200" already pure math; "if spend fell 15%" is hypothetical
    if _HYPOTHETICAL_RE.search(q):
        return True
    return False


def is_general_knowledge_question(question: str) -> bool:
    q = (question or "").strip()
    if not q:
        return False
    if is_document_metadata_question(q) or _DOC_LOOKUP_RE.search(q):
        return False
    if is_statement_field_question(q) or is_statement_identity_question(q):
        return False
    if is_statement_period_question(q) or is_period_coverage_question(q):
        return False
    if is_analyse_overview_ask(q) or is_pdf_overview_question(q):
        return False
    if is_profit_loss_question(q) or is_biggest_expenses_question(q) or is_payment_list_question(q):
        return False
    # Finance-grounded / uploaded-document asks are not general knowledge.
    if re.search(
        r"(?i)\b("
        r"my\s+(?:statement|spend|spending|payments?)|"
        r"how\s+much\s+(?:did|do)\s+i|"
        r"total\s+spent|uploaded\s+(?:pdf|statement)|"
        r"(?:the\s+)?statement\s+(?:period|year|date|balance|summary|holder)|"
        r"account\s+holder|\bifsc\b|branch\s+code"
        r")\b",
        q,
    ):
        return False
    words = len(q.split())
    if words > 28:
        return False
    if _HOWTO_GENERAL_RE.search(q):
        return True
    if _GENERAL_KNOWLEDGE_RE.search(q) or _OFFTOPIC_RE.search(q):
        return True
    if _CONCEPT_LEAD_RE.search(q):
        return True
    return False


def is_conversational_question(question: str) -> bool:
    q = (question or "").strip()
    norm = _normalize_smalltalk(q)
    if not q:
        return True
    if is_self_intro(q) or is_ask_my_name(q) or is_ask_first_question(q) or is_today_date_question(q):
        return True
    from app.saas.personality import is_capability_question, is_identity_question

    if is_identity_question(q) or is_capability_question(q):
        return True
    if _VAGUE_PING_RE.match(norm) or _VAGUE_PING_RE.match(q):
        return True
    if norm in _GREETING_WORDS or norm in _THANKS_WORDS or norm in _BYE_WORDS or norm in _HELP_WORDS:
        return True
    if _HOW_ARE_YOU_RE.match(norm) or _likely_greeting(norm):
        return True
    chat = _detect_intent(q)
    return chat in {"greeting", "thanks", "bye", "help"}


def classify_primary_intent(question: str, *, finance_intent: str | None = None) -> PrimaryRoute:
    """Single pre-retrieval decision for the turn."""
    q = (question or "").strip()
    fin = finance_intent or classify_finance_intent(q)
    chat = _detect_intent(q)
    norm = _normalize_smalltalk(q)

    # 4) Calculations & hypotheticals — before document traps on % / what-if
    if is_pure_math(q) or is_hypothetical_or_calc(q):
        return PrimaryRoute("calculation", "math / hypothetical — no RAG", fin, "math")

    # 5) Conversational
    if is_conversational_question(q):
        return PrimaryRoute("conversational", "greeting / smalltalk / capability", fin, chat)

    # 2) Document metadata — inventory, never content dump
    if is_document_metadata_question(q):
        return PrimaryRoute(
            "document_metadata",
            "KB inventory / filename / pages / status",
            "document_qa",
            "data",
        )

    # Unrelated third-party spend — conversational redirect, not RAG
    if is_unrelated_party_spend_question(q):
        return PrimaryRoute(
            "conversational",
            "unrelated third-party spend",
            "general",
            "offtopic",
        )

    # 3) General knowledge (how-tos, science, concepts) — before document traps
    if is_general_knowledge_question(q):
        return PrimaryRoute("general_knowledge", "base LLM — no document RAG", "general", "concept")

    # 1) Document lookup — statement analytics / fields / overview
    if (
        is_analytical_self_contained(q)
        or is_statement_field_question(q)
        or is_statement_identity_question(q)
        or is_analyse_overview_ask(q)
        or is_pdf_overview_question(q)
        or is_profit_loss_question(q)
        or is_biggest_expenses_question(q)
        or is_payment_list_question(q)
        or is_period_coverage_question(q)
        or is_statement_period_question(q)
        or _DOC_LOOKUP_RE.search(q)
        or (fin not in {"general", "document_qa"} and fin)
    ):
        # document_qa without metadata cues still needs lookup when about "this pdf" content
        return PrimaryRoute(
            "document_lookup",
            "answer depends on uploaded statement contents",
            fin if fin != "general" else "document_qa",
            "data",
        )

    if fin == "document_qa" and not is_document_metadata_question(q):
        return PrimaryRoute("document_lookup", "document content question", fin, "data")

    # Open interrogatives without doc signals → general knowledge
    # (skip lone punctuation / ultra-short vague pings — those are conversational)
    if _VAGUE_PING_RE.match(norm) or _VAGUE_PING_RE.match(q):
        return PrimaryRoute("conversational", "vague ping", "general", "help")
    if re.search(r"(?i)^\s*(what|who|where|when|why|how|which|tell\s+me)\b", q) or (
        q.endswith("?") and len(q.split()) > 1
    ):
        return PrimaryRoute("general_knowledge", "open general question", "general", "concept")

    if _VAGUE_PING_RE.match(norm) or chat == "help":
        return PrimaryRoute("conversational", "clarify / help", "general", "help")

    return PrimaryRoute("conversational", "default conversational", fin, chat)


def primary_uses_retrieval(intent: PrimaryIntent) -> bool:
    """Chroma / statement extract only for content lookup — not metadata inventory."""
    return intent == "document_lookup"


def primary_uses_kb_inventory(intent: PrimaryIntent) -> bool:
    return intent == "document_metadata"
