"""Analyst intent — separate retrieval needs from reasoning.

Six categories (exactly one per turn):
  1. general_chat       — greetings, identity, capability, smalltalk
  2. general_knowledge  — concepts / facts; base LLM only
  3. document_lookup    — factual fields from header/summary (period, holder, year)
  4. financial_analysis — advice / habits / save-cut spend (evidence + LLM)
  5. mixed              — retrieve relevant chunks, then LLM synthesizes
  6. calculations       — pure math / hypotheticals
"""

from __future__ import annotations

import re
from dataclasses import dataclass
from typing import Literal

from app.rag.calculator import is_pure_math
from app.saas.finance_intent import (
    classify_finance_intent,
    is_analyse_overview_ask,
    is_payment_list_question,
    is_period_coverage_question,
    is_profit_loss_question,
    is_statement_field_question,
    is_statement_identity_question,
    is_statement_period_question,
    is_unrelated_party_spend_question,
)
from app.saas.intent_router import (
    is_conversational_question,
    is_document_metadata_question,
    is_general_knowledge_question,
    is_hypothetical_or_calc,
)
from app.saas.knowledge_state import is_kb_meta_question

AnalystIntent = Literal[
    "general_chat",
    "general_knowledge",
    "document_lookup",
    "financial_analysis",
    "mixed",
    "calculations",
]

# Soft user-facing advice / habit questions — need statement evidence + LLM reasoning.
_FINANCIAL_ANALYSIS_RE = re.compile(
    r"(?i)\b("
    r"save\s+money|reduce\s+(?:my\s+)?(?:spend(?:ing)?|expenses?|outflows?)|"
    r"cut\s+(?:back\s+)?(?:on\s+)?(?:spend(?:ing)?|expenses?)|"
    r"spending\s+too\s+much|spend(?:ing)?\s+too\s+much|"
    r"where\s+(?:should|can)\s+i\s+cut|"
    r"(?:analyze|analyse)\s+(?:my\s+)?(?:habits?|spending|spend)|"
    r"spending\s+habits?|budget\s+(?:tips?|advice|help)|"
    r"how\s+(?:can|do)\s+i\s+(?:save|budget|spend\s+less)|"
    r"what\s+should\s+i\s+(?:cut|reduce)|"
    r"overspend(?:ing)?|frugal|money[\s-]?saving|"
    r"advice\s+on\s+(?:my\s+)?(?:spend|money|budget)|"
    r"help\s+me\s+(?:save|budget|cut)|"
    r"am\s+i\s+(?:over)?spend(?:ing)?"
    r")\b"
)

# Factual header/summary asks — do not dump transaction lines.
_FACTUAL_LOOKUP_RE = re.compile(
    r"(?i)\b("
    r"statement\s+(?:year|period|date\s*range)|"
    r"what\s+(?:year|period)|"
    r"account\s+holder|whose\s+statement|"
    r"opening\s+balance|closing\s+balance|"
    r"branch\s*(?:code|name)?|ifsc|micr|customer\s*id|"
    r"bank\s+name|which\s+bank"
    r")\b"
)

# Follow-ups that continue advice without a full re-ask.
_ADVICE_FOLLOW_RE = re.compile(
    r"(?i)^\s*("
    r"reduce\s+(?:it|that|this|them)|"
    r"cut\s+(?:it|that|this)|"
    r"what\s+about\s+(?:this|that|the)\s+merchant|"
    r"and\s+(?:that|this)\s+(?:one|merchant)|"
    r"more\s+(?:tips?|advice|ideas?)|"
    r"how\s+(?:else|otherwise)|"
    r"be\s+more\s+specific"
    r")\b"
)

# Preferred Chroma / content kinds for each retrieval mode.
CHUNK_KINDS_BY_INTENT: dict[str, tuple[str, ...]] = {
    "document_lookup": ("header", "summary", "totals", "statement_period"),
    "financial_analysis": ("totals", "merchant", "transaction", "summary"),
    "mixed": ("header", "summary", "merchant", "transaction", "totals", "statement_period"),
}


@dataclass(frozen=True)
class AnalystRoute:
    intent: AnalystIntent
    reason: str
    needs_document: bool
    needs_llm: bool
    chunk_kinds: tuple[str, ...]
    finance_intent: str


def is_financial_analysis_question(question: str) -> bool:
    return bool(_FINANCIAL_ANALYSIS_RE.search(question or ""))


def is_advice_follow_up(question: str) -> bool:
    return bool(_ADVICE_FOLLOW_RE.search((question or "").strip()))


def is_factual_document_lookup(question: str) -> bool:
    q = (question or "").strip()
    if not q:
        return False
    if is_statement_period_question(q) or is_statement_identity_question(q) or is_statement_field_question(q):
        return True
    if _FACTUAL_LOOKUP_RE.search(q) and not is_financial_analysis_question(q):
        return True
    return False


def classify_analyst_intent(
    question: str,
    *,
    finance_intent: str | None = None,
    prior_was_analysis: bool = False,
) -> AnalystRoute:
    """Pre-retrieval decision: chat / knowledge / lookup / analysis / mixed / calc."""
    q = (question or "").strip()
    fin = finance_intent or classify_finance_intent(q)

    if is_pure_math(q) or is_hypothetical_or_calc(q):
        return AnalystRoute(
            "calculations",
            "math / hypothetical — no document retrieval",
            False,
            True,
            (),
            fin if fin != "general" else "general",
        )

    if is_conversational_question(q) and not is_financial_analysis_question(q):
        return AnalystRoute(
            "general_chat",
            "greeting / identity / capability",
            False,
            True,
            (),
            "general",
        )

    if is_unrelated_party_spend_question(q):
        return AnalystRoute(
            "general_chat",
            "third-party spend redirect",
            False,
            False,
            (),
            "general",
        )

    # Inventory only when not also asking for statement content (name, totals, …).
    if (is_document_metadata_question(q) or is_kb_meta_question(q)) and not (
        is_statement_identity_question(q)
        or is_statement_period_question(q)
        or is_statement_field_question(q)
    ):
        return AnalystRoute(
            "document_lookup",
            "KB inventory / filenames",
            True,
            True,
            CHUNK_KINDS_BY_INTENT["document_lookup"],
            "document_qa",
        )

    if is_financial_analysis_question(q) or (prior_was_analysis and is_advice_follow_up(q)):
        return AnalystRoute(
            "financial_analysis",
            "advice / habits — retrieve evidence then reason",
            True,
            True,
            CHUNK_KINDS_BY_INTENT["financial_analysis"],
            "financial_advice",
        )

    if is_factual_document_lookup(q):
        return AnalystRoute(
            "document_lookup",
            "header/summary factual field",
            True,
            True,
            CHUNK_KINDS_BY_INTENT["document_lookup"],
            fin if fin != "general" else "document_qa",
        )

    # How-to / concepts / science — before any statement path (even if a regex misfired).
    if is_general_knowledge_question(q):
        return AnalystRoute(
            "general_knowledge",
            "base LLM — no document retrieval",
            False,
            True,
            (),
            "general",
        )

    # Analytical statement metrics (spend total, date, merchant list) → mixed:
    # structured evidence preferred; otherwise retrieve + synthesize (never raw dump).
    if (
        fin not in {"general", "document_qa"}
        or is_analyse_overview_ask(q)
        or is_profit_loss_question(q)
        or is_payment_list_question(q)
        or is_period_coverage_question(q)
    ):
        return AnalystRoute(
            "mixed",
            "statement analytics — evidence then natural answer",
            True,
            True,
            CHUNK_KINDS_BY_INTENT["mixed"],
            fin if fin != "general" else "document_qa",
        )

    if is_general_knowledge_question(q) and fin == "general":
        return AnalystRoute(
            "general_knowledge",
            "base LLM — no document retrieval",
            False,
            True,
            (),
            "general",
        )

    if re.search(r"(?i)^\s*(what|who|where|when|why|how|which|tell\s+me)\b", q) or q.endswith("?"):
        return AnalystRoute(
            "general_knowledge",
            "open general question",
            False,
            True,
            (),
            "general",
        )

    return AnalystRoute(
        "general_chat",
        "fallback clarify",
        False,
        True,
        (),
        "general",
    )
