"""Tenant-aware SaaS agent — answers only from each chatbot's uploaded knowledge base."""

from __future__ import annotations

import json
import re
import time
from collections.abc import Callable
from datetime import datetime

from app.config import get_settings
from app.models.schemas import ChatResponse, Citation, StructuredInsight
from app.rag.calculator import calculate as calc_expression
from app.rag.chain import _clean_answer, get_rag_service
from app.rag.prompts import (
    KB_TECH_CONCEPT_PROMPT,
    USER_PROMPT,
)
from app.saas import db as saas_db
from app.saas.assistant_policy import ANALYST_POLICY, suggest_followups
from app.saas.intent_router import is_kb_tech_concept_question
from app.saas.personality import build_concept_prompt, build_meta_prompt, is_capability_question, is_identity_question
from app.saas.chat_router import (
    is_ask_first_question,
    is_ask_my_name,
    is_self_intro,
    route_chat,
    route_summary,
)
from app.saas.analyst_intent import classify_analyst_intent
from app.saas.conversation import (
    get_analysis_context,
    get_first_user_question,
    get_session_display_name,
    get_turns,
    is_analytical_finance_intent,
    mark_session_introduced,
    prior_was_analysis,
    recently_shown_metric,
    remember_analysis_context,
    remember_shown_metrics,
    remember_turn,
    resolve_turn,
    session_was_introduced,
    set_session_display_name,
)
from app.saas.reason import (
    insight_evidence_payload,
    scrub_internal_user_text,
    synthesize_from_evidence,
)
from app.saas.finance_intent import FINANCE_INTENT_LABELS, classify_finance_intent
from app.saas.answer_style import (
    apply_paragraph_style,
    chat_text_from_insight,
    polish_insight_with_llm,
    wants_paragraph_format,
    wants_prose_answer,
)
from app.saas.insight import (
    analysis_evidence_dict,
    build_financial_analysis,
    insight_to_markdown,
    try_structured_kb_answer,
)
from app.saas.kb import (
    agent_document_count,
    retrieve_agent_scored,
)
from app.saas.kb_answer import (
    list_agent_statement_jobs,
    statement_context_notes,
)
from app.saas.knowledge_state import (
    EMPTY_KB_MESSAGE,
    NO_RELEVANT_MESSAGE,
    assess_agent_knowledge,
    unique_ready_docs,
)
from app.saas import rag_debug as dbg
from app.saas.security import AuthContext

AGENT_SYSTEM = """You are {agent_name}, an AI assistant for {org_name} that answers from the uploaded knowledge base.
{persona}

{policy}

Intent for this question: {finance_intent_label}
Conversational mode: {conv_mode}

This turn needs knowledge-base evidence. Evidence notes below are the only document facts you may use — never paste them raw.
- Retrieve → reason → answer. Synthesize a natural reply from the notes.
- Lead with one direct sentence, then brief supporting detail.
- When notes contain a concrete fact (number, date, name, total), state it early.
- Answer narrowly — do not dump entire documents unless the user asked for a full list or summary.
- If this is a follow-up, resolve references using Recent conversation.
- Never invent figures, names, dates, or balances that are not supported by the notes.
- If notes are insufficient, say what is missing — do not guess.
- Attribute naturally: “Based on your uploaded document…” or “From your knowledge base…”.
- Never mention: extraction, incomplete indexing, chunks, OCR, vectors, embeddings, RAG, or internal IDs.

{history}

Evidence notes (most relevant only):
{context}
"""


def _citation_reason(meta: dict, content: str, finance_intent: str) -> str:
    low = (content or "").lower()
    if "total money paid" in low or "payments made" in low:
        return "Statement header totals"
    if finance_intent == "date_lookup" and re.search(r"\b\d{1,2}\s+(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)", low):
        return "Dated payment activity"
    if finance_intent == "category_spending":
        return "Category / merchant narrations"
    if finance_intent == "income" and ("received" in low or "+" in content[:80]):
        return "Credit / inflow lines"
    if "money sent" in low or "paid to" in low:
        return "Payment history detail"
    page = meta.get("page")
    if page is not None:
        return f"Supporting detail from page {page}"
    return "Knowledge base excerpt"


def _format_docs(docs, *, finance_intent: str = "general", scores: list[float] | None = None) -> tuple[str, list[Citation]]:
    blocks: list[str] = []
    citations: list[Citation] = []
    settings = get_settings()
    for idx, doc in enumerate(docs, start=1):
        meta = doc.metadata or {}
        base = settings.max_context_chars_per_doc
        max_chars = max(base, 1800) if meta.get("data_category") == "tenant_kb" else base
        content = (doc.page_content or "")[:max_chars].strip()
        if not content:
            dbg.log_warn("Skipping empty retrieved chunk idx=%s meta=%s", idx, meta)
            continue
        if len(doc.page_content or "") > max_chars:
            content += "..."
        source = meta.get("source_dataset") or meta.get("source_file") or "Statement"
        # Pretty filename for users
        source_name = str(source).split("/")[-1]
        page = meta.get("page")
        chunk_id = meta.get("chunk_id") or meta.get("record_id") or f"src_{idx}"
        score = scores[idx - 1] if scores and idx - 1 < len(scores) else None
        conf = None
        if score is not None:
            conf = max(55, min(98, int(round(50 + float(score) * 100))))
        reason = _citation_reason(meta, content, finance_intent)
        # LLM context: clean, no chunk jargon
        header = f"Source {idx} · {source_name}"
        if page is not None:
            header += f" · page {page}"
        header += f" · {reason}"
        blocks.append(f"{header}:\n{content}")
        citations.append(
            Citation(
                record_id=str(chunk_id),
                data_category=str(meta.get("data_category", "tenant_kb")),
                source_dataset=str(source_name),
                excerpt="",  # do not leak raw OCR to UI
                page=page,
                confidence=conf,
                reason=reason,
            )
        )
    if not blocks:
        dbg.log_error("Context assembly produced zero blocks from %s retrieved docs", len(docs))
    return "\n\n---\n\n".join(blocks), citations


def resolve_kb_mode(auth: AuthContext, agent: dict, override: str | None = None) -> str:
    """Resolve retrieval mode for SaaS agent chat.

    The shared platform finance dataset is NEVER used here — answers come only
    from the selected agent's uploaded knowledge base.
    """
    mode = (override or agent.get("kb_mode") or auth.kb_mode or "tenant").lower()
    if mode not in {"platform", "tenant", "combined"}:
        mode = "tenant"
    has_vectors = agent_document_count(auth.org_id, agent["id"]) > 0
    has_ready = saas_db.count_ready_kb_chunks(auth.org_id, agent["id"]) > 0
    # "platform" historically meant shared dataset — remap to agent-only modes.
    if mode == "platform":
        return "combined" if (has_vectors or has_ready) else "tenant"
    if mode == "combined" and not has_vectors and not has_ready:
        return "tenant"
    return mode


def _greeting_reply(
    *,
    agent: dict,
    agent_name: str,
    org_id: str,
    agent_id: str,
    brief: bool = False,
) -> str:
    """Warm greeting that adapts to whether Knowledge already has files."""
    kb = assess_agent_knowledge(org_id, agent_id)
    if brief:
        if kb.has_knowledge:
            return "Hi — happy to help. What would you like to know about your document?"
        return "Hi! How can I help you today?"

    if kb.has_knowledge:
        docs = unique_ready_docs(
            [
                d
                for d in saas_db.list_kb_documents(org_id, agent_id)
                if d.get("status") == "ready"
            ]
        )
        n = len(docs)
        if n == 1:
            name = docs[0].get("filename") or "your document"
            return (
                f"Hi! I'm {agent_name}.\n\n"
                f"I've indexed **{name}** in your knowledge base.\n\n"
                "Ask me to summarize it, look up facts, or explain anything in the file.\n\n"
                "What would you like to know?"
            )
        if n > 1:
            return (
                f"Hi! I'm {agent_name}.\n\n"
                f"I've got {n} documents ready in your knowledge base. "
                "I can summarize them, find specifics, or compare details across files.\n\n"
                "What would you like to know?"
            )
        return (
            f"Hi! I'm {agent_name}. Your knowledge base is ready — "
            "ask me to summarize, find details, or explain what's in your files."
        )

    custom = (agent.get("welcome_message") or "").strip()
    if custom and "upload a knowledge base first" not in custom.lower():
        return custom
    return (
        f"Hello! I'm {agent_name} — I answer from your uploaded knowledge base. "
        "Upload a PDF, TXT, MD, or CSV in Knowledge, or ask a general question."
    )


def _emit_tokens(text: str, on_token: Callable[[str], None] | None) -> None:
    """Stream precomputed text word-by-word for template / insight replies."""
    if not on_token or not text:
        return
    for piece in re.split(r"(\s+)", text):
        if piece:
            on_token(piece)


def _ollama_text(
    service,
    prompt: str,
    *,
    num_predict: int | None = None,
    on_token: Callable[[str], None] | None = None,
) -> str:
    if on_token:
        parts: list[str] = []
        for chunk in service.stream_ollama(prompt, num_predict=num_predict):
            parts.append(chunk)
            on_token(chunk)
        return "".join(parts)
    return service._invoke_ollama(prompt, num_predict=num_predict)


def agent_chat(
    *,
    question: str,
    auth: AuthContext,
    agent_id: str,
    session_id: str | None = None,
    kb_mode: str | None = None,
    on_token: Callable[[str], None] | None = None,
    on_status: Callable[[str], None] | None = None,
) -> ChatResponse:
    t0 = time.monotonic()
    question = (question or "").strip()
    resolved = resolve_turn(session_id, question)
    # Prefer the user's wording for finance intent; fall back to rewritten follow-up.
    finance_intent = classify_finance_intent(resolved.original)
    if finance_intent == "general" and resolved.rewritten != resolved.original:
        finance_intent = classify_finance_intent(resolved.rewritten)
    effective_q = resolved.rewritten

    agent = saas_db.get_agent(agent_id)

    if not agent or agent["org_id"] != auth.org_id:
        return _finish(
            question,
            "Chatbot not found.",
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent=finance_intent,
        )

    service = get_rag_service()
    mode = resolve_kb_mode(auth, agent, kb_mode)
    agent_name = agent.get("name") or "Assistant"

    # Single routing decision: template / calculator / Ollama / docs(RAG)
    route = route_chat(
        question,
        resolved=resolved,
        finance_intent=finance_intent,
    )
    sid_early = session_id or f"org:{auth.org_id}"
    analyst = classify_analyst_intent(
        question,
        finance_intent=finance_intent,
        prior_was_analysis=prior_was_analysis(sid_early),
    )
    if analyst.intent == "financial_analysis":
        finance_intent = "financial_advice"
        # Force document path so we gather statement evidence then reason.
        from app.saas.chat_router import ChatRoute

        route = ChatRoute(
            "structured_or_rag",
            "data",
            "financial_advice",
            True,
            True,
            analyst.reason,
            "document_lookup",
        )
    chat_intent = route.chat_intent
    finance_intent = route.finance_intent or finance_intent
    dbg.log_info(
        "route target=%s analyst=%s intent=%s finance=%s docs=%s reason=%s q=%r",
        route.target,
        analyst.intent,
        chat_intent,
        finance_intent,
        route.use_docs,
        route.reason,
        question[:120],
    )

    # ── TEMPLATE / CALCULATOR (never RAG) ────────────────────────────────────
    if route.target == "calculator":
        return _finish(
            question,
            calc_expression(question),
            [],
            t0,
            session_id,
            "calculator",
            auth,
            on_token=on_token,
            finance_intent=finance_intent,
            knowledge_state="found",
        )

    if route.target == "template_how_are_you":
        kb = assess_agent_knowledge(auth.org_id, agent_id)
        already = session_was_introduced(session_id) or bool(get_turns(session_id))
        mark_session_introduced(session_id)
        casual = bool(
            re.match(
                r"(?i)^(yo|sup|wass?u+p+|whass?u+p+|(?:what'?s|whats)\s+(?:up|good))[?.!]*$",
                question.strip(),
            )
        )
        if casual and kb.has_knowledge:
            msg = (
                "Hey — all good. What would you like to look at on your statement?"
                if already
                else "Hey — all good. I've got your statement ready — totals, merchants, or a date?"
            )
        elif casual:
            msg = "Hey — all good! Upload a statement in Knowledge, or ask a finance question."
        elif kb.has_knowledge:
            msg = "I'm doing well, thanks! Happy to dig into your statement whenever you're ready."
        else:
            msg = (
                "I'm doing well, thanks! Upload a statement in Knowledge when you're ready, "
                "or ask a general accounting question."
            )
        return _finish(
            question,
            msg,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found" if kb.has_knowledge else "empty",
        )

    if route.target == "template_greeting":
        prior_turns = get_turns(session_id)
        kb = assess_agent_knowledge(auth.org_id, agent_id)
        already = session_was_introduced(session_id) or bool(prior_turns)
        mark_session_introduced(session_id)
        # Full intro at most once per session, and only when there is no doc boot yet.
        use_full = (not already) and (not kb.has_knowledge)
        msg = _greeting_reply(
            agent=agent,
            agent_name=agent_name,
            org_id=auth.org_id,
            agent_id=agent_id,
            brief=not use_full,
        )
        return _finish(
            question,
            msg,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found" if kb.has_knowledge else "empty",
        )

    if route.target == "template_thanks":
        return _finish(
            question,
            "You're welcome — happy to help anytime.",
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found",
        )

    if route.target == "template_bye":
        return _finish(
            question,
            "Goodbye — talk soon.",
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found",
        )

    if route.target == "template_offtopic":
        if "unrelated third-party" in (route.reason or ""):
            offtopic_ans = (
                "I don't have information about that person's spending. "
                "I only analyze statements you upload — ask about your own payments, "
                "merchants, or dates on this statement."
            )
        else:
            offtopic_ans = (
                "I focus on your documents and finances. "
                "Upload a statement in Knowledge, or ask about totals, merchants, or dates."
            )
        return _finish(
            question,
            offtopic_ans,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found",
        )

    if route.target == "template_clarify":
        kb = assess_agent_knowledge(auth.org_id, agent_id)
        soft_ack = bool(
            re.match(
                r"(?i)^(ok|okay|sure|cool|nice|great|alright|fine|got\s+it|k|kk)[?.!]*$",
                question.strip(),
            )
        )
        bare_ping = bool(
            re.match(r"(?i)^(what|huh|eh|and|so|[?!.…]+)[?.!]*$", question.strip())
        )
        if soft_ack and kb.has_knowledge:
            msg = "Got it. What would you like to look at next?"
        elif bare_ping and kb.has_knowledge:
            msg = (
                "Happy to help — ask me to summarize your document, "
                "look up a fact, or explain anything in your knowledge base."
            )
        elif bare_ping:
            msg = (
                "Happy to help — upload a document in Knowledge, "
                "or ask a general question."
            )
        elif kb.has_knowledge:
            msg = (
                "Sure — what should we look at? "
                "I can summarize your knowledge base, find specifics, or explain a section."
            )
        else:
            msg = (
                "I'm not sure what you mean yet. "
                "Upload a document in Knowledge, or ask a general question."
            )
        return _finish(
            question,
            msg,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found" if kb.has_knowledge else "empty",
            show_upload_cta=not kb.has_knowledge,
        )

    if route.target == "template_today":
        today = datetime.now().strftime("%A, %d %B %Y")
        msg = (
            f"Today's date is {today}. "
            "If you meant the statement period instead, ask “what is the statement period?”"
        )
        return _finish(
            question,
            msg,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found",
        )

    if route.target == "template_personal":
        sid = session_id or f"org:{auth.org_id}"
        if is_ask_first_question(question):
            first = get_first_user_question(sid)
            if first:
                msg = f"Your first question was:\n\n“{first}”"
            else:
                msg = "I don't have an earlier question in this chat yet."
        elif is_self_intro(question):
            intro = is_self_intro(question)
            set_session_display_name(sid, intro or "")
            msg = (
                f"Nice to meet you, {intro}! "
                "I'll remember that for this conversation. What would you like to know?"
            )
        elif is_ask_my_name(question):
            known = get_session_display_name(sid)
            holder = None
            jobs = list_agent_statement_jobs(auth.org_id, agent_id)
            if jobs:
                from app.saas.insight import _account_holder

                holder = _account_holder(jobs[0])
            if known and holder and known.lower() not in holder.lower():
                msg = (
                    f"If you mean your name from our conversation, it's {known}. "
                    f"If you mean the account holder on the uploaded statement, it's {holder}."
                )
            elif known:
                msg = f"You told me your name is {known}."
            elif holder:
                # Soft asks like “what do you think is my name” with a statement loaded
                msg = (
                    f"From the uploaded statement, the account holder is {holder}. "
                    "If that’s you, great — or say “my name is …” and I’ll remember it for this chat."
                )
            else:
                msg = (
                    "I don't know your personal name yet — you can say “my name is …”. "
                    "I can also tell you the account holder from an uploaded statement."
                )
        else:
            msg = "Happy to help — what would you like to know?"
        return _finish(
            question,
            msg,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="found",
        )

    if route.target == "template_help":
        kb_status = assess_agent_knowledge(auth.org_id, agent_id)
        sid = session_id or f"org:{auth.org_id}"
        user_name = get_session_display_name(sid)
        history = resolved.history_block or ""
        already = session_was_introduced(sid) or bool(get_turns(sid))
        meta_q = (
            is_identity_question(question)
            or is_capability_question(question)
            or re.search(
                r"(?i)\b(tell\s+me\s+about\s+(?:the\s+bot|this\s+bot|the\s+assistant|alex))\b",
                question,
            )
        )
        if meta_q:
            prompt = build_meta_prompt(
                agent_name=agent_name,
                question=question,
                user_name=user_name,
                has_knowledge=kb_status.has_knowledge,
                has_statement=kb_status.statement_jobs > 0,
                history=history,
                already_spoke=already,
            )
            raw = _ollama_text(
                service,
                prompt,
                num_predict=get_settings().ollama_num_predict_conversational,
                on_token=on_token,
            )
            return _finish(
                question,
                _clean_answer(raw),
                [],
                t0,
                session_id,
                "conversational",
                auth,
                on_token=on_token,
                skip_stream=True,
                finance_intent="general",
                knowledge_state="found" if kb_status.has_knowledge else "empty",
                response_source="general",
            )
        if kb_status.has_knowledge:
            prompt = build_meta_prompt(
                agent_name=agent_name,
                question=question or "What can you help me with?",
                user_name=user_name,
                has_knowledge=True,
                has_statement=kb_status.statement_jobs > 0,
                history=history,
                already_spoke=already,
            )
            raw = _ollama_text(
                service,
                prompt,
                num_predict=get_settings().ollama_num_predict_conversational,
                on_token=on_token,
            )
            return _finish(
                question,
                _clean_answer(raw),
                [],
                t0,
                session_id,
                "conversational",
                auth,
                on_token=on_token,
                skip_stream=True,
                finance_intent="general",
                knowledge_state="found",
                response_source="general",
            )
        return _finish(
            question,
            f"{EMPTY_KB_MESSAGE}\n\n{route_summary()}",
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="general",
            knowledge_state="empty",
            show_upload_cta=True,
        )

    # ── OLLAMA ONLY (concepts / general knowledge / hypotheticals — never RAG) ─
    if route.target == "ollama_concept":
        sid = session_id or f"org:{auth.org_id}"
        user_name = get_session_display_name(sid)
        history = resolved.history_block or ""
        if is_kb_tech_concept_question(question):
            prompt = KB_TECH_CONCEPT_PROMPT.format(question=question)
            if history:
                prompt = f"{history.strip()}\n\n{prompt}"
        else:
            prompt = build_concept_prompt(
                agent_name=agent_name,
                question=question,
                user_name=user_name,
                history=history,
            )
        predict = get_settings().ollama_num_predict_conversational
        if getattr(route, "primary_intent", "") == "calculation":
            predict = max(predict, 120)
        raw = _ollama_text(service, prompt, num_predict=predict, on_token=on_token)
        return _finish(
            question,
            _clean_answer(raw),
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            skip_stream=True,
            finance_intent="general",
            knowledge_state="found",
            response_source="general",
        )

    # ── Document metadata — inventory only (no content RAG / no statement dump) ─
    if route.target == "kb_metadata":
        from app.saas.insight import build_kb_inventory, insight_to_markdown

        inv = build_kb_inventory(auth.org_id, agent_id, question)
        if inv:
            text = insight_to_markdown(inv)
            return _finish(
                question,
                text,
                [],
                t0,
                session_id,
                "pdf",
                auth,
                on_token=on_token,
                finance_intent="document_qa",
                insight=inv,
                knowledge_state="found" if assess_agent_knowledge(auth.org_id, agent_id).has_knowledge else "empty",
                show_upload_cta=not assess_agent_knowledge(auth.org_id, agent_id).has_knowledge,
            )
        return _finish(
            question,
            EMPTY_KB_MESSAGE,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent="document_qa",
            knowledge_state="empty",
            show_upload_cta=True,
        )

    # ── DOCS PATH: structured KB first, then RAG ─────────────────────────────
    kb_status = assess_agent_knowledge(auth.org_id, agent_id)
    data_question = route.use_docs

    # Empty KB: never retrieve shared datasets, never run analysis, never LLM-fabricate.
    if data_question and not kb_status.has_knowledge:
        dbg.log_warn(
            "Empty knowledge for agent=%s vectors=%s ready_docs=%s chunks=%s jobs=%s q=%r",
            agent_id,
            kb_status.vector_count,
            kb_status.ready_docs,
            kb_status.ready_chunks,
            kb_status.statement_jobs,
            question,
        )
        return _finish(
            question,
            EMPTY_KB_MESSAGE,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent=finance_intent,
            knowledge_state="empty",
            show_upload_cta=True,
        )

    # Structured analyst answers — only in hybrid mode (legacy statement cards).
    chat_mode = (get_settings().kb_chat_mode or "fdi").strip().lower()
    use_structured = chat_mode == "hybrid"
    fdi_on = bool(get_settings().fdi_enabled) and chat_mode in {"fdi", "rag"}

    # FDI planner: SQL / identifiers / aggregates before RAG (never invent totals).
    if data_question and kb_status.has_knowledge and fdi_on and chat_mode == "fdi":
        try:
            from app.fdi.planner import run_fdi_turn

            fdi = run_fdi_turn(
                question=question,
                org_id=auth.org_id,
                agent_id=agent_id,
            )
            if fdi.handled and fdi.answer and not fdi.prefer_rag:
                return _finish(
                    question,
                    scrub_internal_user_text(fdi.answer),
                    [],
                    t0,
                    session_id,
                    "pdf",
                    auth,
                    on_token=on_token,
                    finance_intent=finance_intent,
                    knowledge_state="found",
                    response_source="document",
                )
            # prefer_rag with a structured preface — keep answer seed in context later
            if fdi.handled and fdi.answer and fdi.prefer_rag:
                # Stash preface on resolved-like local for context injection
                question_fdi_preface = fdi.answer
            else:
                question_fdi_preface = None
        except Exception as exc:
            dbg.log_warn("FDI planner failed: %s", exc)
            question_fdi_preface = None
    else:
        question_fdi_preface = None

    if data_question and kb_status.has_knowledge and use_structured:
        # Advice follow-ups: reuse prior evidence when possible (no re-retrieval).
        if analyst.intent == "financial_analysis" and prior_was_analysis(sid_early):
            cached = get_analysis_context(sid_early)
            if cached:
                import json as _json

                try:
                    evidence = _json.loads(cached)
                except Exception:
                    evidence = None
                if evidence:
                    user_name = get_session_display_name(sid_early)
                    advice = synthesize_from_evidence(
                        agent_name=agent_name,
                        question=question,
                        evidence=evidence,
                        invoke_ollama=service._invoke_ollama,
                        user_name=user_name,
                        analysis=True,
                        on_token=on_token,
                    )
                    if advice:
                        return _finish(
                            question,
                            scrub_internal_user_text(advice),
                            [],
                            t0,
                            session_id,
                            "pdf",
                            auth,
                            on_token=on_token,
                            skip_stream=True,
                            finance_intent="financial_advice",
                            knowledge_state="found",
                            response_source="statement",
                        )

        structured = try_structured_kb_answer(
            auth.org_id,
            agent_id,
            effective_q,
            original_question=resolved.original,
            conv_intent=resolved.conv_intent,
            prior_question=resolved.prior_question,
            prior_answer=resolved.prior_answer,
        )
        if structured:
            text, insight = structured
            text, insight, streamed = _finalize_structured_answer(
                insight,
                question,
                service,
                session_id=session_id,
                on_token=on_token,
                agent_name=agent_name,
                force_analysis=analyst.intent == "financial_analysis"
                or insight.intent == "financial_advice",
            )
            if insight.intent == "financial_advice":
                import json as _json

                remember_analysis_context(
                    sid_early,
                    _json.dumps(analysis_evidence_dict(insight, question), ensure_ascii=False),
                )
            return _finish(
                question,
                scrub_internal_user_text(text),
                [],
                t0,
                session_id,
                "pdf",
                auth,
                on_token=on_token,
                skip_stream=streamed,
                finance_intent=insight.intent or finance_intent,
                insight=insight,
                knowledge_state="found",
                response_source="statement",
            )

        # Financial analysis with a statement job but no structured hit — build evidence.
        if analyst.intent == "financial_analysis":
            jobs = list_agent_statement_jobs(auth.org_id, agent_id)
            if jobs:
                insight = build_financial_analysis(jobs[0], question)
                text, insight, streamed = _finalize_structured_answer(
                    insight,
                    question,
                    service,
                    session_id=session_id,
                    on_token=on_token,
                    agent_name=agent_name,
                    force_analysis=True,
                )
                import json as _json

                remember_analysis_context(
                    sid_early,
                    _json.dumps(analysis_evidence_dict(insight, question), ensure_ascii=False),
                )
                return _finish(
                    question,
                    scrub_internal_user_text(text),
                    [],
                    t0,
                    session_id,
                    "pdf",
                    auth,
                    on_token=on_token,
                    skip_stream=streamed,
                    finance_intent="financial_advice",
                    insight=insight,
                    knowledge_state="found",
                    response_source="statement",
                )

        # Analytical / challenge questions must not fall through to chunk-only RAG
        # when a statement job exists — ask for a clearer metric instead.
        if (
            is_analytical_finance_intent(finance_intent)
            or resolved.conv_intent in {"verify", "explain_calc"}
            or (
                resolved.conv_intent == "follow_up"
                and resolved.prior_question
                and is_analytical_finance_intent(classify_finance_intent(resolved.prior_question))
            )
        ):
            jobs = list_agent_statement_jobs(auth.org_id, agent_id)
            if jobs:
                answer = (
                    "I re-checked the full statement but still need a clearer ask to compute that precisely. "
                    "Try: statement period · how much did I spend · spend on 16 July · groceries · money received."
                )
                return _finish(
                    question,
                    answer,
                    [],
                    t0,
                    session_id,
                    "pdf",
                    auth,
                    on_token=on_token,
                    finance_intent=finance_intent,
                    knowledge_state="found",
                )
            # No statement job — continue to vector retrieval over this agent's indexed docs.

    # Retrieval query: bias short follow-ups with prior question
    retrieval_q = effective_q
    if (
        resolved.conv_intent == "follow_up"
        and resolved.prior_question
        and len(resolved.original.split()) <= 12
    ):
        retrieval_q = f"{resolved.prior_question}\n{resolved.original}"

    docs = []
    retrieval_reason = None
    scores: list[float] = []

    # Only this agent's uploaded documents — never the shared platform dataset.
    if kb_status.has_knowledge:
        if on_status:
            on_status("searching")
        settings = get_settings()
        chat_mode = (settings.kb_chat_mode or "fdi").strip().lower()
        overviewish = bool(
            re.search(r"(?i)\b(analy[sz]e|summar(?:y|ize)|overview|about)\b", question or "")
        )
        top_k = settings.kb_retrieval_top_k if overviewish else max(4, min(6, settings.kb_retrieval_top_k))
        # Restrict statement chunk kinds only in legacy hybrid mode.
        kinds = None
        if chat_mode == "hybrid":
            kinds = analyst.chunk_kinds or None
            if analyst.intent == "document_lookup":
                kinds = ("header", "summary", "totals", "statement_period")
        if settings.fdi_hybrid_search and chat_mode in ("fdi", "rag"):
            from app.fdi.hybrid import hybrid_retrieve

            scored, retrieval_reason = hybrid_retrieve(
                auth.org_id,
                agent_id,
                retrieval_q,
                k=top_k,
                chunk_kinds=kinds,
            )
        else:
            scored, retrieval_reason = retrieve_agent_scored(
                auth.org_id,
                agent_id,
                retrieval_q,
                k=top_k,
                chunk_kinds=kinds,
            )
        docs.extend([d for d, _ in scored])
        scores = [s for _, s in scored]

    if not docs:
        dbg.log_error(
            "No retrieval context for query=%r mode=%s kb=%s reason=%s",
            question,
            mode,
            kb_status,
            retrieval_reason,
        )
        if not kb_status.has_knowledge:
            return _finish(
                question,
                EMPTY_KB_MESSAGE,
                [],
                t0,
                session_id,
                "conversational",
                auth,
                on_token=on_token,
                finance_intent=finance_intent,
                knowledge_state="empty",
                show_upload_cta=True,
            )
        return _finish(
            question,
            NO_RELEVANT_MESSAGE,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent=finance_intent,
            knowledge_state="no_relevant",
        )

    context, citations = _format_docs(docs, finance_intent=finance_intent, scores=scores)
    if not context.strip():
        dbg.log_error("Empty context after formatting; refusing to call LLM with blank notes")
        return _finish(
            question,
            NO_RELEVANT_MESSAGE,
            [],
            t0,
            session_id,
            "conversational",
            auth,
            on_token=on_token,
            finance_intent=finance_intent,
            knowledge_state="no_relevant",
        )

    if mode in {"tenant", "combined"} and kb_status.has_knowledge:
        chat_mode = (get_settings().kb_chat_mode or "fdi").strip().lower()
        # Hybrid mode may prepend statement totals; FDI/RAG use retrieved chunks.
        if chat_mode == "hybrid":
            stmt = statement_context_notes(auth.org_id, agent_id)
            if stmt:
                stmt_clean = (
                    stmt.replace("=== SAMPLE TRANSACTIONS", "=== TRANSACTION SAMPLES")
                    .replace("=== BANK STATEMENT SUMMARY", "=== STATEMENT TOTALS")
                    .replace("(prefer these totals for cashflow questions)", "")
                )
                context = stmt_clean + "\n\n---\n\n" + context
        if question_fdi_preface:
            context = (
                "=== STRUCTURED FACTS (verified; prefer these numbers) ===\n"
                + question_fdi_preface
                + "\n\n---\n\n"
                + context
            )
        if len(docs) < 3:
            context = (
                context
                + "\n\nNote to assistant: retrieval coverage may be incomplete for this question; "
                "say if the answer may be incomplete. "
                "Do not claim you reviewed records that are not in the notes. "
                "Never invent totals — use structured facts when present."
            )

    persona = ""
    if agent.get("description"):
        persona = f"Persona / role: {agent['description']}"
    history = resolved.history_block or ""
    conv_mode = {
        "new_question": "new question",
        "follow_up": "follow-up — resolve references using conversation history",
        "verify": "challenge — re-check data, do not defend prior answer",
        "explain_calc": "explain calculation from full statement",
    }.get(resolved.conv_intent, "new question")

    prompt = (
        AGENT_SYSTEM.format(
            agent_name=agent_name,
            org_name=auth.org_name,
            persona=persona,
            policy=ANALYST_POLICY,
            finance_intent_label=FINANCE_INTENT_LABELS.get(finance_intent, "General"),
            conv_mode=conv_mode,
            history=history,
            context=context,
        )
        + "\n\n"
        + USER_PROMPT.format(question=resolved.original)
    )
    dbg.print_rag_debug(
        query=question,
        docs=docs,
        scores=scores if scores else None,
        prompt=prompt,
    )
    answer = _clean_answer(_ollama_text(service, prompt, on_token=on_token))
    answer = scrub_internal_user_text(_scrub_internal_language(answer))
    return _finish(
        question,
        answer,
        citations,
        t0,
        session_id,
        "rag",
        auth,
        on_token=on_token,
        skip_stream=True,
        finance_intent=finance_intent,
        knowledge_state="found",
        response_source="document",
    )


def _scrub_internal_language(text: str) -> str:
    """Last-line defense against model leaking retrieval jargon / false review claims."""
    if not text:
        return text
    replacements = [
        (r"(?i)\bextracted\s+payments?\b", "payments"),
        (r"(?i)\bextracted\s+(?:line\s*)?items?\b", "line items"),
        (r"(?i)\bfrom\s+the\s+statement\s+text\b", "on the statement"),
        (r"(?i)\b(?:vector|embedding|chunk|OCR|RAG)\b", ""),
        (r"(?i)\b\d+\s+chunks?\b", ""),
        (
            r"(?i)\bi\s+reviewed\s+your\s+(?:uploaded\s+)?(?:documents?|financial\s+records?|records?|statements?)\b",
            "Looking at your documents",
        ),
        (r"[ \t]{2,}", " "),
    ]
    out = text
    for pat, repl in replacements:
        out = re.sub(pat, repl, out)
    return out.strip()


def _dedupe_insight_metrics(
    insight: StructuredInsight,
    session_id: str | None,
) -> StructuredInsight:
    """Drop metric cards that were already shown this session (keep narrative + txns)."""
    if not insight or not insight.metrics or not session_id:
        return insight
    # Always keep lean ranked lists (biggest expenses) intact.
    if insight.headline and "biggest" in insight.headline.lower():
        remember_shown_metrics(
            session_id,
            [(m.label, m.value) for m in insight.metrics if m.label and m.value],
        )
        return insight

    kept = []
    for m in insight.metrics:
        if m.label and m.value and recently_shown_metric(session_id, m.label, m.value):
            # Skip total spent / received / net / payment counts already shown.
            if re.search(
                r"(?i)^(total spent|total received|net cashflow|payments made|"
                r"payments received|transactions|account holder|period|date range)$",
                m.label.strip(),
            ):
                continue
        kept.append(m)
    if kept != list(insight.metrics):
        insight = insight.model_copy(update={"metrics": kept})
    remember_shown_metrics(
        session_id,
        [(m.label, m.value) for m in (insight.metrics or []) if m.label and m.value],
    )
    return insight


def _finalize_structured_answer(
    insight: StructuredInsight,
    question: str,
    service,
    *,
    session_id: str | None = None,
    on_token: Callable[[str], None] | None = None,
    agent_name: str = "Alex",
    force_analysis: bool = False,
) -> tuple[str, StructuredInsight, bool]:
    """Grounded facts first; optional LLM prose for natural phrasing."""
    streamed = False
    if wants_paragraph_format(question) and insight.transactions and not force_analysis:
        insight = apply_paragraph_style(insight, question)
        insight = _dedupe_insight_metrics(insight, session_id)
        return scrub_internal_user_text(chat_text_from_insight(insight)), insight, streamed

    # Financial advice / habits — always reason over evidence with the LLM.
    if force_analysis or insight.intent == "financial_advice":
        user_name = get_session_display_name(session_id)
        evidence = analysis_evidence_dict(insight, question)
        advice = synthesize_from_evidence(
            agent_name=agent_name,
            question=question,
            evidence=evidence,
            invoke_ollama=service._invoke_ollama,
            user_name=user_name,
            analysis=True,
            on_token=on_token,
        )
        insight = _dedupe_insight_metrics(insight, session_id)
        if advice:
            insight = insight.model_copy(update={"summary": advice})
            return scrub_internal_user_text(advice), insight, bool(on_token)
        return scrub_internal_user_text(chat_text_from_insight(insight)), insight, streamed

    use_llm = wants_prose_answer(question) or (
        insight.intent == "statement_summary"
        and not insight.transactions
        and re.search(r"(?i)\banaly[sz]e|overview|summar|tell\s+me\s+about|what\s+do\s+you\s+think", question or "")
    )
    if use_llm:
        polished = synthesize_from_evidence(
            agent_name=agent_name,
            question=question,
            evidence=insight_evidence_payload(insight),
            invoke_ollama=service._invoke_ollama,
            user_name=get_session_display_name(session_id),
            analysis=False,
            on_token=on_token,
        )
        if not polished:
            polished = polish_insight_with_llm(
                insight,
                question,
                invoke_ollama=lambda prompt, num_predict=None: _ollama_text(
                    service, prompt, num_predict=num_predict, on_token=on_token
                ),
            )
        if polished:
            insight = insight.model_copy(update={"summary": polished})
            streamed = bool(on_token)

    insight = _dedupe_insight_metrics(insight, session_id)
    return scrub_internal_user_text(chat_text_from_insight(insight)), insight, streamed


def _kb_help_answer(org_id: str, agent_id: str, agent_name: str) -> str | None:
    docs = unique_ready_docs(
        [
            d
            for d in saas_db.list_kb_documents(org_id, agent_id)
            if d.get("status") == "ready"
        ]
    )
    if not docs:
        return None
    names = [d.get("filename") or "document" for d in docs[:5]]
    listed = ", ".join(f"“{n}”" for n in names)
    extra = f" and {len(docs) - 5} more" if len(docs) > 5 else ""
    # Avoid PDF re-extract here — filename suffix is enough for the tip line.
    has_statement = any(str(d.get("filename") or "").lower().endswith(".pdf") for d in docs)
    lines = [
        f"I'm {agent_name}. I can answer from {listed}{extra}.",
    ]
    if has_statement:
        lines.append(
            "I can summarize the statement, totals, merchants, dates, or list payments — what would you like?"
        )
    else:
        lines.append("Ask about those documents, or upload a statement in Knowledge.")
    return " ".join(lines)


def _finish(
    question: str,
    answer: str,
    citations: list[Citation],
    t0: float,
    session_id: str | None,
    mode: str,
    auth: AuthContext,
    *,
    finance_intent: str | None = None,
    insight: StructuredInsight | None = None,
    knowledge_state: str | None = None,
    show_upload_cta: bool = False,
    on_token: Callable[[str], None] | None = None,
    skip_stream: bool = False,
    response_source: str | None = None,
) -> ChatResponse:
    if on_token and not skip_stream:
        _emit_tokens(answer, on_token)
    from app import database

    duration_ms = int((time.monotonic() - t0) * 1000)
    sid = session_id or f"org:{auth.org_id}"
    mark_session_introduced(sid)
    if response_source is None:
        if mode == "rag":
            response_source = "document"
        elif mode in {"pdf", "pdf_overview"} or insight is not None:
            response_source = "statement"
        else:
            response_source = "general"
    has_docs = knowledge_state == "found" or bool(insight)
    has_statement = bool(
        insight
        and insight.intent
        in {
            "statement_summary",
            "spending_summary",
            "date_lookup",
            "income",
            "payment_count",
            "category_spending",
            "merchant_analysis",
            "transaction_search",
            "period_coverage",
        }
    )
    suggestions = suggest_followups(
        finance_intent=finance_intent,
        insight=insight,
        has_statement=has_statement
        or (knowledge_state == "found" and mode in {"pdf", "rag", "conversational"}),
        has_docs=has_docs and not show_upload_cta,
    )
    if insight is not None and not insight.suggestions:
        insight = insight.model_copy(update={"suggestions": suggestions})
    payload = {
        "citations": [c.model_dump() for c in citations],
        "finance_intent": finance_intent,
        "insight": insight.model_dump() if insight else None,
        "knowledge_state": knowledge_state,
        "suggestions": suggestions,
    }
    citations_json = json.dumps(payload)
    log_id = database.log_query(question, answer, citations_json, duration_ms, sid)
    remember_turn(sid, question, answer)
    return ChatResponse(
        answer=answer,
        citations=citations,
        session_id=sid,
        log_id=log_id or None,
        mode=mode if mode in {"conversational", "rag", "pdf", "pdf_overview", "calculator"} else "rag",
        pdf_job_id=None,
        pdf_summary=None,
        finance_intent=finance_intent,
        insight=insight,
        knowledge_state=knowledge_state,  # type: ignore[arg-type]
        show_upload_cta=show_upload_cta,
        suggestions=suggestions,
        response_source=response_source,  # type: ignore[arg-type]
    )
