"""Build analyst-style structured insights from statement jobs — never expose retrieval jargon."""

from __future__ import annotations

import re
from typing import Any

from app.models.schemas import (
    InsightMetric,
    InsightTable,
    InsightTransaction,
    StructuredInsight,
)
from app.pdf.calc import (
    _detect_currency_symbol,
    _detect_spend_category,
    _money_out,
    _money_in,
    _pretty_money,
    _rough_category,
    _row_matches_keys,
    is_pdf_overview_question,
)
from app.saas.answer_style import apply_paragraph_style, wants_paragraph_format
from app.saas.conversation import is_coverage_challenge
from app.saas.finance_intent import (
    FINANCE_INTENT_LABELS,
    classify_finance_intent,
    extract_merchant_query,
    is_analyse_overview_ask,
    is_biggest_expenses_question,
    is_financial_advice_question,
    is_payment_list_question,
    is_period_coverage_question,
    is_profit_loss_question,
    is_statement_field_question,
    is_statement_identity_question,
    is_statement_period_question,
    is_vague_single_payment_question,
)
from app.saas.kb_answer import (
    _MONTH_ALIASES,
    _date_sort_key,
    _full_statement_text,
    _month_token,
    _normalize_date_label,
    _parse_day_month,
    _row_matches_day_month,
    _spend_by_date_from_rows,
    _spend_by_date_from_text,
    is_per_date_spend_question,
    list_agent_statement_jobs,
)
from app.saas import db as saas_db
from app.saas.knowledge_state import unique_ready_docs

from datetime import datetime, timezone, timedelta

_MONTH_FULL = {
    "jan": "January",
    "feb": "February",
    "mar": "March",
    "apr": "April",
    "may": "May",
    "jun": "June",
    "jul": "July",
    "aug": "August",
    "sep": "September",
    "oct": "October",
    "nov": "November",
    "dec": "December",
}
_MONTH_ORDER = list(_MONTH_FULL.keys())


def _cur(job: dict[str, Any]) -> str:
    return _detect_currency_symbol(job) or "₹"


def _money(v: object, cur: str) -> str:
    return _pretty_money(v, cur) or f"{cur} 0.00"


def _clean_desc(text: str, limit: int = 56) -> str:
    t = re.sub(r"\s+", " ", (text or "").replace("\n", " ")).strip()
    t = re.sub(r"(?i)\b(upi\s*id|upi\s*ref\s*no|tag:|note:).*$", "", t).strip(" -|,")
    if not t:
        return "Payment"
    return t if len(t) <= limit else t[: limit - 1].rstrip() + "…"


def _header_metrics(job: dict[str, Any]) -> dict[str, Any]:
    calc = job.get("calculation") or {}
    metrics = dict(calc.get("metrics") or {})
    header = job.get("statement_totals") or {}
    for k in ("money_out", "money_in", "payments_made", "payments_received", "net_cashflow"):
        if metrics.get(k) is None and header.get(k) is not None:
            metrics[k] = header.get(k)
    # When header totals are missing (common on HDFC extracts), derive from rows.
    out_rows = _outflow_rows(job)
    in_rows = _inflow_rows(job)
    if metrics.get("money_out") is None and out_rows:
        metrics["money_out"] = round(sum(_money_out(r) for r in out_rows), 2)
    if metrics.get("money_in") is None and in_rows:
        metrics["money_in"] = round(sum(_money_in(r) for r in in_rows), 2)
    if metrics.get("payments_made") is None and out_rows:
        metrics["payments_made"] = len(out_rows)
    if metrics.get("payments_received") is None and in_rows:
        metrics["payments_received"] = len(in_rows)
    if metrics.get("net_cashflow") is None and metrics.get("money_in") is not None and metrics.get("money_out") is not None:
        try:
            metrics["net_cashflow"] = float(metrics["money_in"]) - float(metrics["money_out"])
        except (TypeError, ValueError):
            pass
    return metrics


def _period_label(job: dict[str, Any]) -> str | None:
    text = _full_statement_text(job)[:2500]
    m = re.search(
        r"(?i)(\d{1,2}\s*(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)'?\d{0,4}"
        r"\s*[-–]\s*\d{1,2}\s*(?:JAN|FEB|MAR|APR|MAY|JUN|JUL|AUG|SEP|OCT|NOV|DEC)'?\d{0,4})",
        text,
    )
    if m:
        return re.sub(r"\s+", " ", m.group(1)).strip()
    # HDFC-style: Statement From : 01/05/2026 To: 31/05/2026
    m2 = re.search(
        r"(?i)statement\s+from\s*[:\-]?\s*(\d{1,2}[/-]\d{1,2}[/-]\d{2,4})"
        r"\s*to\s*[:\-]?\s*(\d{1,2}[/-]\d{1,2}[/-]\d{2,4})",
        text,
    )
    if m2:
        return f"{m2.group(1)} - {m2.group(2)}"
    return None


def _account_holder(job: dict[str, Any]) -> str | None:
    """Best-effort account / statement holder from header text (bank-agnostic)."""
    text = _full_statement_text(job)[:2500]
    labeled = re.search(
        r"(?i)(?:customer\s*name|account\s*holder|account\s*name|name\s*of\s*customer)\s*[:\-]\s*"
        r"([A-Z][A-Za-z]+(?:\s+[A-Z][A-Za-z]+){0,3})",
        text,
    )
    if labeled:
        return re.sub(r"\s+", " ", labeled.group(1)).strip()
    # HDFC / many bank PDFs: "MR NITIN KUMAR" (optionally followed by state/city noise)
    honorific = re.search(
        r"(?i)\b((?:mr|mrs|ms|smt|shri)\.?\s+[A-Z][A-Za-z]+(?:\s+[A-Z][A-Za-z]+){0,3})\b",
        text,
    )
    if honorific:
        name = re.sub(r"\s+", " ", honorific.group(1)).strip()
        # Drop trailing geography tokens wrongly glued on
        name = re.sub(
            r"(?i)\s+(HARYANA|PUNJAB|DELHI|INDIA|CITY|STATE)\b.*$",
            "",
            name,
        ).strip()
        if len(name.split()) >= 2:
            return name.title() if name.isupper() else name
    # Paytm / many UPI PDFs: ALL-CAPS name near the top
    for line in text.splitlines():
        s = line.strip()
        if not s or s.startswith("---"):
            continue
        if re.match(r"^page\s*\d+", s, re.I):
            continue
        if re.match(r"^[A-Z][A-Z\s.'-]{2,48}$", s):
            skip = (
                "PAYTM",
                "STATEMENT",
                "HDFC",
                "ICICI",
                "AXIS",
                "SBI",
                "BANK",
                "UPI",
                "NOTE",
                "PANCHKULA",
                "HARYANA",
                "SECTOR",
                "ADDRESS",
                "JOINT",
            )
            if any(tok in s for tok in skip):
                continue
            return re.sub(r"\s+", " ", s).title()
        # Title Case name on its own line
        if re.match(r"^[A-Z][a-z]+(?:\s+[A-Z][a-z]+){1,3}$", s) and "@" not in s:
            return s
    return None


def _extract_statement_fields(job: dict[str, Any]) -> dict[str, str]:
    """Pull common bank-header identifiers from statement text."""
    text = _full_statement_text(job)[:4000]
    fields: dict[str, str] = {}

    def _grab(key: str, pattern: str) -> None:
        m = re.search(pattern, text, re.I)
        if m:
            val = re.sub(r"\s+", " ", m.group(1)).strip(" :-")
            if val:
                fields[key] = val

    _grab(
        "account_branch",
        r"account\s+branch\s*[:\-]?\s*([A-Za-z0-9][A-Za-z0-9 /,&-]{1,40})",
    )
    # HDFC often puts city on the next line after Account Branch
    if fields.get("account_branch"):
        m_city = re.search(
            r"(?i)account\s+branch\s*[:\-]?\s*[^\n]+\n\s*([A-Z][A-Za-z ]{2,30})",
            text,
        )
        if m_city:
            city = m_city.group(1).strip()
            if city.upper() not in {"ADDRESS", "JOINT HOLDERS", "CURRENCY"}:
                fields["account_branch"] = f"{fields['account_branch']}, {city}"

    _grab("branch_code", r"branch\s*code\s*[:\-]?\s*(\d{3,8})")
    _grab(
        "account_number",
        r"(?:account\s*no\.?|a/?c\s*no\.?|account\s*number)\s*[:\-]?\s*(\d{9,18})",
    )
    _grab("ifsc", r"(?:rtgs/?neft\s*)?ifsc\s*[:\-]?\s*([A-Z]{4}0[A-Z0-9]{6})")
    _grab("micr", r"micr\s*[:\-]?\s*(\d{6,12})")
    _grab("cust_id", r"cust(?:omer)?\s*id\s*[:\-]?\s*(\d{5,15})")
    _grab("state", r"\bstate\s*:\s*([A-Z][A-Za-z ]{2,30})")
    # Bank name — prefer explicit "HDFC Bank Ltd" style headers
    _grab(
        "bank_name",
        r"\b((?:HDFC|ICICI|SBI|AXIS|YES|KOTAK|IDFC|PNB|BOB|CANARA|UNION)\s+Bank(?:\s+Ltd\.?)?)\b",
    )
    if not fields.get("bank_name"):
        _grab(
            "bank_name",
            r"\b([A-Z][A-Za-z&. ]{2,40}\s+Bank(?:\s+Ltd\.?)?)\b",
        )
    if fields.get("state"):
        fields["state"] = re.split(
            r"(?i)\b(?:phone|email|od\s*limit|currency|india)\b",
            fields["state"],
        )[0].strip(" .")
        fields["state"] = " ".join(fields["state"].split()[:3]).strip(" ,.")

    holder = _account_holder(job)
    if holder:
        fields["account_holder"] = holder
    period = _period_label(job)
    if period:
        fields["period"] = period
    return fields


def build_statement_field_answer(job: dict[str, Any], question: str) -> StructuredInsight:
    """Answer branch / IFSC / account number style asks from the statement header."""
    fields = _extract_statement_fields(job)
    filename = job.get("filename") or "statement"
    q = (question or "").lower()

    # Map question → preferred field keys (ordered).
    wanted: list[tuple[str, str]] = []
    # Ambiguous compound ask first (e.g. "state account branch number")
    if re.search(r"(?i)state\s+account\s+branch|statement\s+account\s+branch", q):
        wanted = [
            ("Branch code", "branch_code"),
            ("Account branch", "account_branch"),
            ("Account number", "account_number"),
            ("IFSC", "ifsc"),
            ("State", "state"),
        ]
    else:
        if re.search(r"(?i)\bifsc\b", q):
            wanted.append(("IFSC", "ifsc"))
        if re.search(r"(?i)\bmicr\b", q):
            wanted.append(("MICR", "micr"))
        if re.search(r"(?i)\bcust(?:omer)?\s*id\b", q):
            wanted.append(("Customer ID", "cust_id"))
        if re.search(r"(?i)\baccount\s*(?:number|no\.?|num)\b|\ba/?c\s*(?:number|no)", q):
            wanted.append(("Account number", "account_number"))
        if re.search(r"(?i)\bbranch\s*code\b|\bbranch\s*(?:number|no\.?|num)\b", q):
            wanted.append(("Branch code", "branch_code"))
            wanted.append(("Account branch", "account_branch"))
        if re.search(r"(?i)\baccount\s+branch\b", q):
            wanted.append(("Account branch", "account_branch"))
            wanted.append(("Branch code", "branch_code"))
        if re.search(r"(?i)\bstate\b", q) and not re.search(r"(?i)\bstatement\b", q):
            wanted.append(("State", "state"))
        if re.search(r"(?i)\bbranch\b", q) and not wanted:
            wanted = [
                ("Branch code", "branch_code"),
                ("Account branch", "account_branch"),
                ("Account number", "account_number"),
                ("IFSC", "ifsc"),
                ("State", "state"),
            ]
        if re.search(r"(?i)\bbank\s+name\b|\bname\s+of\s+(?:the\s+)?bank\b|\bwhich\s+bank\b|\bwhat\s+bank\b", q):
            wanted = [("Bank", "bank_name"), ("IFSC", "ifsc"), ("Account branch", "account_branch")] + wanted

    # Deduplicate while preserving order
    seen: set[str] = set()
    ordered: list[tuple[str, str]] = []
    for label, key in wanted:
        if key in seen:
            continue
        seen.add(key)
        ordered.append((label, key))

    metrics: list[InsightMetric] = []
    for label, key in ordered:
        if fields.get(key):
            metrics.append(InsightMetric(label=label, value=fields[key], tone="accent"))

    if not metrics:
        # Nothing matched the ask — still show whatever header fields we found.
        for label, key in (
            ("Branch code", "branch_code"),
            ("Account branch", "account_branch"),
            ("Account number", "account_number"),
            ("IFSC", "ifsc"),
            ("State", "state"),
            ("Customer ID", "cust_id"),
        ):
            if fields.get(key):
                metrics.append(InsightMetric(label=label, value=fields[key], tone="neutral"))

    if fields.get("account_holder"):
        metrics.append(
            InsightMetric(label="Account holder", value=fields["account_holder"], tone="neutral")
        )
    if fields.get("period"):
        metrics.append(InsightMetric(label="Period", value=fields["period"], tone="neutral"))

    if not metrics:
        return StructuredInsight(
            intent="statement_summary",
            intent_label=FINANCE_INTENT_LABELS["statement_summary"],
            headline="Statement details",
            summary=(
                f"I couldn’t find branch / account identifiers on “{filename}”. "
                "Try asking for the statement overview, or check the first page of the PDF."
            ),
            metrics=[],
            footnotes=["Looked through the statement header text."],
        )

    primary = metrics[0]
    summary = (
        f"From “{filename}”: {primary.label} is {primary.value}."
        if len(metrics) == 1
        else f"Here are the account / branch details from “{filename}”."
    )
    return StructuredInsight(
        intent="statement_summary",
        intent_label=FINANCE_INTENT_LABELS["statement_summary"],
        headline="Account & branch details",
        summary=summary,
        metrics=metrics,
        footnotes=["Read from the statement header / first page — not a general definition."],
    )


def build_cashflow_verdict(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    """Focused profit/loss style answer from statement net cashflow — not a full dump."""
    cur = _cur(job)
    m = _header_metrics(job)
    filename = job.get("filename") or "statement"
    period = _period_label(job)
    holder = _account_holder(job)
    received = m.get("money_in")
    spent = m.get("money_out")
    net = m.get("net_cashflow")
    if net is None and received is not None and spent is not None:
        try:
            net = float(received) - float(spent)
        except (TypeError, ValueError):
            net = None

    if net is None:
        return StructuredInsight(
            intent="insights",
            intent_label=FINANCE_INTENT_LABELS["insights"],
            headline="Cashflow",
            summary=(
                f"I couldn’t determine net cashflow on “{filename}”. "
                "Ask for total spent and total received, or upload a clearer statement."
            ),
            metrics=[],
            footnotes=[],
        )

    net_f = float(net)
    if net_f < 0:
        verdict = (
            f"Based on this statement, money out exceeded money in — "
            f"net cashflow is {_money(net_f, cur)} (a net outflow for the period)."
        )
        tone = "danger"
        label = "Net outflow"
    elif net_f > 0:
        verdict = (
            f"Based on this statement, money in exceeded money out — "
            f"net cashflow is {_money(net_f, cur)} (a net inflow for the period)."
        )
        tone = "success"
        label = "Net inflow"
    else:
        verdict = "Based on this statement, inflows and outflows balanced — net cashflow is zero."
        tone = "neutral"
        label = "Net cashflow"

    note = (
        "This is statement cashflow (credits vs debits for the period), "
        "not accounting profit/loss or your live account balance."
    )
    metrics = [
        InsightMetric(label=label, value=_money(net_f, cur), tone=tone),
    ]
    if spent is not None:
        metrics.append(InsightMetric(label="Total spent", value=_money(spent, cur), tone="danger"))
    if received is not None:
        metrics.append(InsightMetric(label="Total received", value=_money(received, cur), tone="success"))
    if period:
        metrics.append(InsightMetric(label="Period", value=period, tone="neutral"))
    if holder:
        metrics.append(InsightMetric(label="Account holder", value=holder, tone="neutral"))

    return StructuredInsight(
        intent="insights",
        intent_label=FINANCE_INTENT_LABELS["insights"],
        headline="Profit or loss on this statement?",
        summary=verdict + " " + note,
        metrics=metrics,
        highlights=[],
        footnotes=[f"Figures from “{filename}” statement totals."],
    )


def _payment_blocks(job: dict[str, Any]) -> list[list[str]]:
    """Split statement text into date-headed payment blocks (Paytm-style)."""
    text = _full_statement_text(job)
    date_start = re.compile(
        r"(?i)^(\d{1,2})\s+(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\b"
    )
    lines = text.splitlines()
    blocks: list[list[str]] = []
    cur: list[str] = []
    for line in lines:
        if date_start.match(line.strip()):
            if cur:
                blocks.append(cur)
            cur = [line]
        elif cur:
            cur.append(line)
    if cur:
        blocks.append(cur)
    return blocks


def _block_to_row(block: list[str]) -> dict[str, Any] | None:
    date_start = re.compile(
        r"(?i)^(\d{1,2})\s+(Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)\b"
    )
    months = r"Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec"
    paid_re = re.compile(
        r"(?i)(?:Paid to|Money sent to|Received from|Recharge of|Transferred to)\s+(.+?)"
        r"(?:\s+Note:|\s+UPI\s+ID:|\s+Tag:|\s+HDFC|\s+payu|\s+State Bank|\s*$)"
    )
    blob = "\n".join(block)
    dm = date_start.match(block[0].strip())
    if not dm:
        return None
    flat = re.sub(r"\s+", " ", blob)
    # Skip statement period headers: "3 JUL'26 - 2 AUG'26 - Rs.30,298.30"
    if re.search(
        rf"(?i)\d{{1,2}}\s+(?:{months})\S*\s*[-–]\s*\d{{1,2}}\s+(?:{months})",
        flat,
    ):
        return None
    # Require a real payment cue — avoid header totals / page chrome.
    if not re.search(
        r"(?i)\b("
        r"paid\s+to|money\s+sent|received\s+from|recharge\s+of|transferred\s+to|"
        r"upi\s+ref|money\s+received"
        r")\b",
        flat,
    ):
        return None
    amt = None
    m_out = re.search(r"(?i)-\s*-?\s*(?:Rs\.?|₹)\s*([\d,]+\.?\d*)", blob)
    if m_out:
        try:
            amt = float(m_out.group(1).replace(",", ""))
        except ValueError:
            amt = None
    if not amt or amt <= 0:
        return None
    paid = paid_re.search(flat)
    if paid:
        prefix = paid.group(0).split(paid.group(1))[0].strip()
        desc = f"{prefix} {paid.group(1).strip()}".strip()
    else:
        desc = re.sub(r"\s+", " ", block[0])[:80]
    desc = re.sub(r"\s+(HDFC|ICICI|Axis|SBI|Bank).*$", "", desc, flags=re.I).strip()
    date_label = f"{int(dm.group(1)):02d} {dm.group(2)[:3].title()}"
    return {
        "date": date_label,
        "description": desc[:80],
        "debit": amt,
        "credit": None,
        "amount": -amt,
        "raw": blob[:240],
        "_from_text": True,
        "_day": int(dm.group(1)),
        "_month": dm.group(2)[:3].lower(),
    }


def _day_hits_from_text(job: dict[str, Any], day: int, month: str) -> list[dict[str, Any]]:
    mon = (month or "")[:3].lower()
    hits: list[dict[str, Any]] = []
    for block in _payment_blocks(job):
        row = _block_to_row(block)
        if not row:
            continue
        if row.get("_day") == day and row.get("_month") == mon:
            hits.append(row)
    return hits


def _month_hits_from_text(job: dict[str, Any], month: str) -> list[dict[str, Any]]:
    mon = (month or "")[:3].lower()
    hits: list[dict[str, Any]] = []
    for block in _payment_blocks(job):
        row = _block_to_row(block)
        if not row:
            continue
        if row.get("_month") == mon:
            hits.append(row)
    return hits


def _outflow_rows(job: dict[str, Any]) -> list[dict[str, Any]]:
    """Table-extracted outflows only (may be incomplete vs statement header)."""
    return [r for r in (job.get("rows") or []) if _money_out(r) > 0]


def _inflow_rows(job: dict[str, Any]) -> list[dict[str, Any]]:
    return [r for r in (job.get("rows") or []) if _money_in(r) > 0]


def _text_outflow_rows(job: dict[str, Any]) -> list[dict[str, Any]]:
    """Outflows parsed from full statement text blocks (Paytm-style)."""
    hits: list[dict[str, Any]] = []
    for block in _payment_blocks(job):
        row = _block_to_row(block)
        if row and _money_out(row) > 0:
            hits.append(row)
    return hits


def _complete_outflow_rows(job: dict[str, Any]) -> list[dict[str, Any]]:
    """Union of table rows + text blocks — same source month/date lookups use.

    Keeps “list all payments” consistent with August/July month queries.
    """
    return _dedupe_category_rows(_text_outflow_rows(job) + _outflow_rows(job))


def _header_payments_made(job: dict[str, Any]) -> int | None:
    m = _header_metrics(job)
    raw = m.get("payments_made")
    if raw is None:
        return None
    try:
        return int(raw)
    except (TypeError, ValueError):
        return None


def _extraction_gap(job: dict[str, Any], indexed_n: int | None = None) -> tuple[int, int] | None:
    """Return (indexed, header) when indexed count is short of the statement header."""
    header_n = _header_payments_made(job)
    if header_n is None:
        return None
    n = indexed_n if indexed_n is not None else len(_complete_outflow_rows(job))
    if n < header_n:
        return (n, header_n)
    return None


_FULL_LIST_RE = re.compile(
    r"(?i)\b("
    r"all|every|each|full|complete|entire|list|list\s+out|show\s+(?:me\s+)?all|"
    r"break\s*down|detail(?:ed|s)?|itemi[sz]e|every\s+payment|all\s+payments|"
    r"all\s+transactions?|full\s+list|only\s+(?:a\s+)?few|more\s+(?:payments?|transactions?|rows?)|"
    r"rest\s+of\s+(?:the\s+)?(?:list|payments?|transactions?)|show\s+(?:everything|more)"
    r")\b"
)
_TXN_HARD_CAP = 500


def _wants_full_list(question: str) -> bool:
    return bool(_FULL_LIST_RE.search(question or ""))


def _wants_payment_count(question: str) -> bool:
    return bool(re.search(r"(?i)\bhow\s+many\s+(?:payments?|transactions?|txns?)\b", question or ""))


def _row_date_label(row: dict[str, Any]) -> str:
    return (
        _normalize_date_label(str(row.get("date") or ""))
        or str(row.get("date") or "").split("\n")[0][:12]
    )


def _txn_items(
    rows: list[dict[str, Any]],
    *,
    out: bool = True,
    limit: int | None = None,
    full: bool = False,
    sort_by_date: bool = False,
) -> list[InsightTransaction]:
    """Build transaction cards. Prefer full chronological lists for analytical answers."""
    if full or limit is None:
        cap = _TXN_HARD_CAP
        sort_by_date = True
    else:
        cap = max(1, min(int(limit), _TXN_HARD_CAP))

    if sort_by_date:
        ordered = sorted(rows, key=lambda r: _date_sort_key(_row_date_label(r) or ""))
    else:
        ordered = sorted(
            rows,
            key=lambda r: _money_out(r) if out else _money_in(r),
            reverse=True,
        )

    items: list[InsightTransaction] = []
    for r in ordered[:cap]:
        amt = _money_out(r) if out else _money_in(r)
        date = _row_date_label(r)
        items.append(
            InsightTransaction(
                date=date or None,
                description=_clean_desc(str(r.get("description") or "Payment"), limit=72),
                amount=f"{amt:,.2f}",
                direction="out" if out else "in",
            )
        )
    return items


def _list_footnote(
    shown: int,
    indexed_total: int,
    *,
    label: str = "payments",
    header_n: int | None = None,
) -> str | None:
    """Honest list coverage — never claim 'all' when header count is higher."""
    if indexed_total <= 0 and not header_n:
        return None
    if header_n is not None and shown < header_n:
        return (
            f"Showing {shown} of {header_n} {label} from the statement for this view — "
            "ask about a date or merchant for more detail."
        )
    if shown < indexed_total:
        return (
            f"Showing {shown} of {indexed_total} {label}. "
            "Ask for the full list if you need every payment."
        )
    if header_n is not None and shown >= header_n:
        return f"Showing all {shown} {label} reported on the statement."
    return f"Showing all {shown} {label} retrieved from the statement."


def _extraction_incomplete_blurb(job: dict[str, Any], indexed_n: int | None = None) -> str | None:
    """Internal gap detector — do not surface extraction jargon to users."""
    return None


def _coverage_footnotes(job: dict[str, Any], *, matched_n: int | None = None) -> list[str]:
    """User-facing footnotes — no extraction/indexing internals."""
    notes: list[str] = []
    if matched_n is not None and matched_n > 0:
        notes.append("Figures are based on the payments found on your uploaded statement.")
    return notes


def build_financial_analysis(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    """Evidence pack for save/cut/habit questions — metrics + merchant concentration."""
    cur = _cur(job)
    m = _header_metrics(job)
    out_rows = _complete_outflow_rows(job) or _outflow_rows(job)
    q = (question or "").strip()

    total = m.get("money_out")
    count = m.get("payments_made")
    if total is None and out_rows:
        total = round(sum(_money_out(r) for r in out_rows), 2)
        count = len(out_rows)
    period = _period_label(job)
    avg = None
    if total is not None and count:
        try:
            avg = float(total) / max(int(count), 1)
        except (TypeError, ValueError):
            avg = None

    # Focused merchant / category named in the advice question (e.g. Zomato).
    focus_name: str | None = None
    focus_keys: tuple[str, ...] = ()
    cat = _detect_spend_category(q) if q else None
    if cat:
        focus_name, focus_keys = cat[0], cat[1]
    else:
        merchant = extract_merchant_query(q) if q else None
        if merchant:
            focus_name = merchant
            focus_keys = (merchant.lower(),)
            tokens = [t for t in re.split(r"\s+", merchant.lower()) if len(t) > 2]
            if tokens and tokens[0] not in focus_keys:
                focus_keys = (merchant.lower(), tokens[0])

    focus_matched: list[dict[str, Any]] = []
    focus_total = 0.0
    focus_count = 0
    if focus_name and focus_keys:
        text_hits = _category_hits_from_text(job, focus_keys)
        rows = [r for r in (job.get("rows") or []) if _money_out(r) > 0 and _row_matches_keys(r, focus_keys)]
        focus_matched = _dedupe_category_rows(text_hits + rows) if text_hits or rows else []
        focus_total = round(sum(_money_out(r) for r in focus_matched), 2)
        focus_count = len(focus_matched)

    by_merchant: dict[str, dict[str, float | int]] = {}
    for r in out_rows:
        desc = _clean_desc(str(r.get("description") or "")) or "Unknown"
        key = " ".join(desc.split()[:3]) if desc else "Unknown"
        bucket = by_merchant.setdefault(key, {"spend": 0.0, "count": 0})
        bucket["spend"] = float(bucket["spend"]) + _money_out(r)
        bucket["count"] = int(bucket["count"]) + 1

    ranked = sorted(by_merchant.items(), key=lambda x: -float(x[1]["spend"]))
    top = ranked[:5]
    highest = sorted(out_rows, key=_money_out, reverse=True)[:5]

    display = (focus_name or "").title() if focus_name and focus_name.islower() else (focus_name or "")
    metrics: list[InsightMetric] = []
    if focus_name:
        share = None
        if total and float(total) > 0 and focus_count:
            share = 100.0 * focus_total / float(total)
        metrics.append(
            InsightMetric(
                label=f"{display} spend",
                value=_money(focus_total, cur) if focus_count else "₹ 0.00",
                hint=f"{focus_count} payments" + (f" · {share:.0f}% of total" if share is not None else ""),
                tone="danger" if focus_count else "neutral",
            )
        )
    metrics.extend(
        [
            InsightMetric(label="Total spent", value=_money(total, cur), tone="danger"),
            InsightMetric(label="Payments", value=str(int(count or 0)), tone="neutral"),
        ]
    )
    if avg is not None:
        metrics.append(InsightMetric(label="Average spend", value=_money(avg, cur), tone="accent"))
    if period:
        metrics.append(InsightMetric(label="Period", value=period, tone="neutral"))
    if top and total and float(total) > 0 and not focus_name:
        top_name, top_stats = top[0]
        share = 100.0 * float(top_stats["spend"]) / float(total)
        metrics.append(
            InsightMetric(
                label="Top merchant share",
                value=f"{share:.0f}% · {top_name}",
                hint=f"{int(top_stats['count'])} payments",
                tone="accent",
            )
        )

    highlights: list[str] = []
    if focus_name and focus_count:
        share_bit = ""
        if total and float(total) > 0:
            share_bit = f" ({100.0 * focus_total / float(total):.0f}% of spend)"
        highlights.append(
            f"{display}: {_money(focus_total, cur)} across {focus_count} payments{share_bit}"
        )
        for r in sorted(focus_matched, key=_money_out, reverse=True)[:3]:
            highlights.append(
                f"{display} payment: {_money(_money_out(r), cur)} · {_clean_desc(str(r.get('description') or ''))}"
            )
    elif focus_name:
        highlights.append(
            f"No clear {display} payments found on this statement — advice below uses overall concentration."
        )
    for name, stats in top[:3]:
        pct = ""
        if total and float(total) > 0:
            pct = f" ({100.0 * float(stats['spend']) / float(total):.0f}% of spend)"
        highlights.append(
            f"{name}: {_money(float(stats['spend']), cur)} across {int(stats['count'])} payments{pct}"
        )
    for r in highest[:3]:
        highlights.append(
            f"Largest: {_money(_money_out(r), cur)} · {_clean_desc(str(r.get('description') or ''))}"
        )

    sample_rows = focus_matched if focus_matched else highest
    txns = _format_txn_amounts(_txn_items(sample_rows, out=True, full=False, limit=8), cur)

    recommendations: list[str] = []
    if focus_name and focus_count:
        recommendations.append(
            f"Start with {display} — {_money(focus_total, cur)} across {focus_count} payments."
        )
        if focus_count >= 4:
            recommendations.append(
                f"Set a weekly {display} cap and skip 1–2 orders to cut that bill meaningfully."
            )
    elif top:
        name, stats = top[0]
        recommendations.append(
            f"Start with {name} — highest concentration ({int(stats['count'])} payments)."
        )
    if avg is not None and highest and not (focus_name and focus_count):
        hi = _money_out(highest[0])
        if hi > avg * 2:
            recommendations.append(
                f"One payment of {_money(hi, cur)} is well above your average of {_money(avg, cur)}."
            )
    if len(top) >= 2 and not (focus_name and focus_count):
        recommendations.append(
            f"Next, trim repeat spend at {top[1][0]} ({int(top[1][1]['count'])} payments)."
        )

    if focus_name and focus_count:
        summary = (
            f"On this statement you spent {_money(focus_total, cur)} on {display} "
            f"across {focus_count} payments. "
            + " ".join(recommendations[:2])
        )
        headline = f"Cutting {display} spend"
    else:
        summary = (
            "Based on your uploaded statement, here is where spending concentrates "
            "and a few practical places to cut."
        )
        if recommendations:
            summary = summary + " " + " ".join(recommendations[:2])
        headline = "Spending habits & savings ideas"

    return StructuredInsight(
        intent="financial_advice",
        intent_label=FINANCE_INTENT_LABELS.get("financial_advice", "Financial Advice"),
        headline=headline,
        summary=summary,
        metrics=metrics,
        highlights=highlights[:8],
        transactions=txns,
        footnotes=[],
        suggestions=[
            "How much did I spend overall?",
            "Show my biggest expenses",
            f"How much spent on {display}?" if display else "How much spent on groceries?",
        ],
    )


def analysis_evidence_dict(insight: StructuredInsight, question: str | None = None) -> dict[str, Any]:
    """Compact evidence JSON for the LLM advice pass."""
    payload: dict[str, Any] = {
        "headline": insight.headline,
        "summary": insight.summary,
        "metrics": [{"label": m.label, "value": m.value, "hint": m.hint} for m in insight.metrics],
        "highlights": list(insight.highlights or []),
        "sample_payments": [
            {"date": t.date, "description": t.description, "amount": t.amount}
            for t in (insight.transactions or [])[:10]
        ],
        "recommendations_seed": list(insight.highlights or [])[:4],
    }
    q = (question or "").strip()
    focus_name: str | None = None
    if q:
        cat = _detect_spend_category(q)
        if cat:
            focus_name = cat[0]
        else:
            focus_name = extract_merchant_query(q)
    if not focus_name:
        # Recover from metrics like "Zomato spend"
        for m in insight.metrics or []:
            label = (m.label or "").strip()
            if re.search(r"(?i)\bspend\b", label) and not re.search(
                r"(?i)^(total|average)\s+spend", label
            ):
                focus_name = re.sub(r"(?i)\s+spend\s*$", "", label).strip() or None
                if focus_name:
                    payload["focus_merchant"] = {
                        "name": focus_name,
                        "total": m.value,
                        "detail": m.hint,
                        "found": True,
                    }
                    break
    elif focus_name:
        display = focus_name.title() if focus_name.islower() else focus_name
        focus_metric = next(
            (m for m in (insight.metrics or []) if focus_name.lower() in (m.label or "").lower()),
            None,
        )
        found = False
        payments = None
        if focus_metric and focus_metric.hint:
            m_count = re.search(r"(\d+)\s+payments?", focus_metric.hint)
            if m_count:
                payments = int(m_count.group(1))
                found = payments > 0
        payload["focus_merchant"] = {
            "name": display,
            "total": focus_metric.value if focus_metric else None,
            "detail": focus_metric.hint if focus_metric else None,
            "payments": payments,
            "found": found,
        }
    return payload


def _format_txn_amounts(items: list[InsightTransaction], cur: str) -> list[InsightTransaction]:
    out = []
    for t in items:
        raw = t.amount.replace(",", "")
        try:
            val = float(raw)
            amt = _money(val, cur)
        except ValueError:
            amt = t.amount
        out.append(t.model_copy(update={"amount": amt or t.amount}))
    return out


def insight_to_markdown(insight: StructuredInsight) -> str:
    """Fallback plain answer for copy / audit — still analyst tone, no jargon."""
    lines = [insight.headline, "", insight.summary]
    if insight.metrics:
        lines.append("")
        for m in insight.metrics:
            hint = f" ({m.hint})" if m.hint else ""
            lines.append(f"- **{m.label}:** {m.value}{hint}")
    if insight.highlights:
        lines.append("")
        lines.append("**Highlights**")
        for h in insight.highlights:
            lines.append(f"- {h}")
    if insight.table and insight.table.headers and insight.table.rows:
        lines.append("")
        lines.append("| " + " | ".join(insight.table.headers) + " |")
        lines.append("| " + " | ".join("---" for _ in insight.table.headers) + " |")
        for row in insight.table.rows:
            lines.append("| " + " | ".join(row) + " |")
    if insight.transactions:
        lines.append("")
        n = len(insight.transactions)
        lines.append(f"**Payments ({n})**" if n > 8 else "**Payments**")
        for t in insight.transactions:
            when = f"{t.date} · " if t.date else ""
            lines.append(f"- {when}{t.description} — {t.amount}")
    if insight.footnotes:
        lines.append("")
        for f in insight.footnotes:
            lines.append(f"_{f}_")
    return "\n".join(lines).strip()


def _overview_narrative(job: dict[str, Any]) -> str:
    """Insightful overview prose — observations, not a restatement of metric cards."""
    cur = _cur(job)
    m = _header_metrics(job)
    period = _period_label(job)
    holder = _account_holder(job)
    out_rows = _outflow_rows(job)
    largest = max(out_rows, key=_money_out) if out_rows else None

    spent = m.get("money_out")
    received = m.get("money_in")
    net = m.get("net_cashflow")
    if net is None and spent is not None and received is not None:
        try:
            net = float(received) - float(spent)
        except (TypeError, ValueError):
            net = None

    parts: list[str] = []
    who = f" for {holder}" if holder else ""
    when = f" ({period})" if period else ""

    if net is not None:
        net_f = float(net)
        if net_f < 0:
            parts.append(
                f"Looking at this statement{who}{when}, money going out ran ahead of money coming in — "
                f"a net outflow of {_money(net_f, cur)}."
            )
        elif net_f > 0:
            parts.append(
                f"Looking at this statement{who}{when}, inflows beat outflows — "
                f"a net inflow of {_money(net_f, cur)}."
            )
        else:
            parts.append(
                f"Looking at this statement{who}{when}, inflows and outflows roughly balanced."
            )
    else:
        parts.append(f"Here’s a quick read on this statement{who}{when}.")

    if largest:
        parts.append(
            f"The standout outflow was {_money(_money_out(largest), cur)} "
            f"({_clean_desc(str(largest.get('description') or 'payment'))})."
        )

    made = m.get("payments_made")
    if spent is not None and made:
        try:
            avg = float(spent) / max(int(made), 1)
            parts.append(
                f"Across {int(made)} payments made, the average send was about {_money(avg, cur)}."
            )
        except (TypeError, ValueError, ZeroDivisionError):
            pass

    recv_n = m.get("payments_received")
    if made and recv_n is not None:
        try:
            if int(made) > max(int(recv_n), 1) * 3:
                parts.append(
                    "Outgoing activity is much denser than credits — everyday spend is driving the picture."
                )
        except (TypeError, ValueError):
            pass

    parts.append(
        "The cards show the raw totals — the takeaway is the cashflow direction and "
        "what is driving it, not just the headline numbers."
    )
    return " ".join(parts)


def build_statement_summary(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    cur = _cur(job)
    m = _header_metrics(job)
    filename = job.get("filename") or "Your statement"
    period = _period_label(job)
    holder = _account_holder(job)
    out_rows = _outflow_rows(job)
    largest = max(out_rows, key=_money_out) if out_rows else None
    full = _wants_full_list(question or "")
    q = question or ""

    # Focused answers for identity / period / header-field asks
    if is_statement_field_question(q):
        return build_statement_field_answer(job, q)

    if is_statement_identity_question(q):
        if holder:
            return StructuredInsight(
                intent="statement_summary",
                intent_label=FINANCE_INTENT_LABELS["statement_summary"],
                headline="Account holder",
                summary=f"This statement belongs to {holder}.",
                metrics=[
                    InsightMetric(label="Account holder", value=holder, tone="accent"),
                    *([InsightMetric(label="Period", value=period, tone="neutral")] if period else []),
                    InsightMetric(label="Document", value=str(filename), tone="neutral"),
                ],
                highlights=[],
                footnotes=["Read from the statement header / first page."],
            )
        return StructuredInsight(
            intent="statement_summary",
            intent_label=FINANCE_INTENT_LABELS["statement_summary"],
            headline="Account holder",
            summary=(
                f"I couldn’t find a clear account-holder name on “{filename}”. "
                "Try opening the first page of the PDF, or ask for the statement overview."
            ),
            metrics=[],
            highlights=[],
            footnotes=[],
        )

    if is_statement_period_question(q) and period:
        return StructuredInsight(
            intent="statement_summary",
            intent_label=FINANCE_INTENT_LABELS["statement_summary"],
            headline="Statement period",
            summary=f"This statement covers {period}.",
            metrics=[
                InsightMetric(label="Period", value=period, tone="accent"),
                *([InsightMetric(label="Account holder", value=holder, tone="neutral")] if holder else []),
            ],
            highlights=[],
            footnotes=["Taken from the statement header date range."],
        )

    metrics = []
    if holder:
        metrics.append(InsightMetric(label="Account holder", value=holder, tone="accent"))
    if m.get("money_out") is not None:
        metrics.append(InsightMetric(label="Total spent", value=_money(m["money_out"], cur), tone="danger"))
    if m.get("money_in") is not None:
        metrics.append(InsightMetric(label="Total received", value=_money(m["money_in"], cur), tone="success"))
    if m.get("net_cashflow") is not None:
        net = float(m["net_cashflow"])
        metrics.append(
            InsightMetric(
                label="Net cashflow",
                value=_money(net, cur),
                tone="danger" if net < 0 else "success",
            )
        )
    if m.get("payments_made") is not None:
        metrics.append(InsightMetric(label="Payments made", value=str(int(m["payments_made"])), tone="neutral"))
    if m.get("payments_received") is not None:
        metrics.append(InsightMetric(label="Payments received", value=str(int(m["payments_received"])), tone="neutral"))
    if period:
        metrics.append(InsightMetric(label="Period", value=period, tone="neutral"))

    highlights = []
    if largest:
        highlights.append(
            f"Largest outflow: {_money(_money_out(largest), cur)} — {_clean_desc(str(largest.get('description') or ''))}"
        )
    if m.get("money_out") and m.get("payments_made"):
        try:
            avg = float(m["money_out"]) / max(int(m["payments_made"]), 1)
            highlights.append(f"Average payment size: {_money(avg, cur)}")
        except (TypeError, ValueError, ZeroDivisionError):
            pass

    txns = []
    if full:
        txns = _format_txn_amounts(
            _txn_items(out_rows, out=True, full=True, limit=None),
            cur,
        )
    footnotes = []
    if full:
        footnotes.append("Ask about spend by date, a merchant, or a category for a deeper cut.")
        note = _list_footnote(len(txns), len(out_rows))
        if note:
            footnotes.insert(0, note)
    else:
        footnotes.append("Ask for payment details, a date, or a merchant for a deeper cut.")

    if full:
        summary = (
            f"Here’s a clean overview of “{filename}”"
            + (f" for {holder}" if holder else "")
            + (f" covering {period}" if period else "")
            + ". Figures in the cards come from the statement totals."
        )
    else:
        # Observations / insights — cards already show the raw totals.
        summary = _overview_narrative(job)

    return StructuredInsight(
        intent="statement_summary",
        intent_label=FINANCE_INTENT_LABELS["statement_summary"],
        headline="Statement overview",
        summary=summary,
        metrics=metrics,
        highlights=highlights,
        transactions=txns,
        footnotes=footnotes,
    )


def build_biggest_expenses(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    """Ranked outflows with merchant, amount, and date — not a totals restatement."""
    cur = _cur(job)
    q = question or ""
    top_n = 10
    m_top = re.search(r"(?i)\btop\s+(\d+)\b", q)
    if m_top:
        try:
            top_n = max(3, min(25, int(m_top.group(1))))
        except ValueError:
            top_n = 10

    out_rows = _complete_outflow_rows(job)
    ranked = sorted(out_rows, key=_money_out, reverse=True)[:top_n]
    txns = _format_txn_amounts(
        [
            InsightTransaction(
                date=_row_date_label(r) or None,
                description=_clean_desc(str(r.get("description") or "Payment"), limit=72),
                amount=f"{_money_out(r):,.2f}",
                direction="out",
            )
            for r in ranked
        ],
        cur,
    )

    if not txns:
        return StructuredInsight(
            intent="insights",
            intent_label=FINANCE_INTENT_LABELS["insights"],
            headline="Biggest expenses",
            summary="I couldn’t rank individual expenses — no indexed outflow lines on this statement yet.",
            metrics=[],
            transactions=[],
            footnotes=_coverage_footnotes(job),
        )

    top = ranked[0]
    share = ""
    header_spent = _header_metrics(job).get("money_out")
    if header_spent:
        try:
            pct = 100.0 * _money_out(top) / float(header_spent)
            share = f" That’s about {pct:.0f}% of total statement spend on its own."
        except (TypeError, ValueError, ZeroDivisionError):
            share = ""

    incomplete = _extraction_incomplete_blurb(job, len(out_rows))
    summary = (
        f"Here are your top {len(txns)} expenses by amount — each with merchant, amount, and date. "
        f"The largest is {_money(_money_out(top), cur)} to "
        f"{_clean_desc(str(top.get('description') or 'a payee'))}"
        f"{(' on ' + (_row_date_label(top) or '').strip()) if _row_date_label(top) else ''}."
        f"{share}"
    )
    if incomplete:
        summary = f"{summary} {incomplete}"

    # Lean metrics — avoid repeating the full statement totals card set.
    metrics = [
        InsightMetric(label="Shown", value=str(len(txns)), tone="accent"),
        InsightMetric(
            label="Largest",
            value=_money(_money_out(top), cur) or "",
            tone="danger",
        ),
    ]

    table = InsightTable(
        headers=["#", "Date", "Merchant / description", "Amount"],
        rows=[
            [
                str(i + 1),
                t.date or "—",
                t.description or "Payment",
                t.amount or "",
            ]
            for i, t in enumerate(txns)
        ],
    )

    return StructuredInsight(
        intent="insights",
        intent_label=FINANCE_INTENT_LABELS["insights"],
        headline="Biggest expenses",
        summary=summary,
        metrics=metrics,
        highlights=[],
        transactions=txns,
        table=table,
        footnotes=_coverage_footnotes(job)
        + ["Ranked from indexed outflows (statement text + table rows), largest first."],
    )


def build_spending_summary(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    cur = _cur(job)
    m = _header_metrics(job)
    out_rows = _complete_outflow_rows(job) if _wants_full_list(question or "") else _outflow_rows(job)
    total = m.get("money_out")
    count = m.get("payments_made")
    if total is None and out_rows:
        total = round(sum(_money_out(r) for r in out_rows), 2)
        count = len(out_rows)
    period = _period_label(job)
    full = _wants_full_list(question or "")
    avg = None
    if total is not None and count:
        try:
            avg = float(total) / max(int(count), 1)
        except (TypeError, ValueError):
            avg = None

    metrics = [
        InsightMetric(label="Total spent", value=_money(total, cur), tone="danger"),
        InsightMetric(label="Transactions", value=str(int(count or 0)), tone="neutral"),
    ]
    if avg is not None:
        metrics.append(InsightMetric(label="Average", value=_money(avg, cur), tone="accent"))
    if period:
        metrics.append(InsightMetric(label="Date range", value=period, tone="neutral"))

    highlights = []
    top = sorted(out_rows, key=_money_out, reverse=True)[:3]
    for r in top:
        highlights.append(f"{_money(_money_out(r), cur)} · {_clean_desc(str(r.get('description') or ''))}")

    table = None
    txns: list = []
    if full:
        cats: dict[str, float] = {}
        for r in out_rows:
            cat = _rough_category(str(r.get("description") or ""))
            cats[cat] = cats.get(cat, 0.0) + _money_out(r)
        if cats:
            table = InsightTable(
                headers=["Category", "Amount"],
                rows=[
                    [name.title(), _money(amt, cur) or ""]
                    for name, amt in sorted(cats.items(), key=lambda x: -x[1])
                ],
            )
        txns = _format_txn_amounts(_txn_items(out_rows, out=True, full=True), cur)

    footnotes: list[str] = []
    header_n = _header_payments_made(job)
    if full:
        footnotes.append("Totals prefer the statement header when available.")
        note = _list_footnote(len(txns), len(out_rows), header_n=header_n)
        if note:
            footnotes.insert(0, note)
        footnotes.extend(_coverage_footnotes(job))

    if full:
        incomplete = _extraction_incomplete_blurb(job, len(txns))
        if incomplete:
            summary = (
                f"{incomplete} Listing the {len(txns)} indexed outflow"
                f"{'s' if len(txns) != 1 else ''} I could retrieve."
            )
        elif txns:
            summary = (
                f"Here’s how money left the account — "
                f"{len(txns)} indexed outflow{'s' if len(txns) != 1 else ''} listed below."
            )
        else:
            summary = "Here’s how money left the account on this statement, with the largest outflows highlighted."
    else:
        # Direct answer first; cards hold the same totals — keep supporting detail short.
        top_note = ""
        if top:
            top_note = (
                f" Largest was {_money(_money_out(top[0]), cur)} "
                f"({_clean_desc(str(top[0].get('description') or 'payment'))})."
            )
        summary = (
            f"You spent {_money(total, cur)} across {int(count or 0)} payment"
            f"{'s' if int(count or 0) != 1 else ''} on this statement.{top_note}"
        )

    return StructuredInsight(
        intent="spending_summary",
        intent_label=FINANCE_INTENT_LABELS["spending_summary"],
        headline="Spending summary",
        summary=summary,
        metrics=metrics,
        highlights=highlights,
        transactions=txns,
        table=table,
        footnotes=footnotes,
    )


def build_income_summary(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    cur = _cur(job)
    m = _header_metrics(job)
    in_rows = _inflow_rows(job)
    total = m.get("money_in")
    count = m.get("payments_received")
    if total is None and in_rows:
        total = round(sum(_money_in(r) for r in in_rows), 2)
        count = len(in_rows)
    period = _period_label(job)
    metrics = [
        InsightMetric(label="Total received", value=_money(total, cur), tone="success"),
        InsightMetric(label="Credits", value=str(int(count or 0)), tone="neutral"),
    ]
    if period:
        metrics.append(InsightMetric(label="Date range", value=period, tone="accent"))
    txns = _format_txn_amounts(_txn_items(in_rows, out=False, full=True), cur)
    footnotes = []
    note = _list_footnote(len(txns), len(in_rows), label="credits")
    if note:
        footnotes.append(note)
    return StructuredInsight(
        intent="income",
        intent_label=FINANCE_INTENT_LABELS["income"],
        headline="Money received",
        summary="Credits and inflows recorded on this statement.",
        metrics=metrics,
        transactions=txns,
        highlights=[],
        footnotes=footnotes,
    )


def build_date_lookup(job: dict[str, Any], question: str) -> StructuredInsight:
    cur = _cur(job)
    filename = job.get("filename") or "statement"

    if is_per_date_spend_question(question):
        by_text = _spend_by_date_from_text(_full_statement_text(job))
        by_rows = _spend_by_date_from_rows(job.get("rows") or [])
        by = by_text if len(by_text) >= len(by_rows) else by_rows
        if not by:
            return StructuredInsight(
                intent="date_lookup",
                intent_label=FINANCE_INTENT_LABELS["date_lookup"],
                headline="Spend by date",
                summary=(
                    f"I reviewed dated outflows in “{filename}” but couldn’t assemble a reliable day-by-day view. "
                    "Try a specific day such as “spend on 16 July”."
                ),
                metrics=[],
                highlights=[],
                footnotes=["Searched payment history lines with calendar dates on the statement."],
            )
        total = round(sum(float(v["total"]) for v in by.values()), 2)
        n = sum(int(v["n"]) for v in by.values())
        # Full day-by-day table — every date with spend
        table = InsightTable(
            headers=["Date", "Spent", "Payments"],
            rows=[
                [label, _money(by[label]["total"], cur) or "", str(int(by[label]["n"]))]
                for label in sorted(by.keys(), key=_date_sort_key)
            ],
        )
        # Also attach every indexed outflow for the full payment list (same set as “list all”)
        out_rows = _complete_outflow_rows(job)
        txns = _format_txn_amounts(_txn_items(out_rows, out=True, full=True), cur)
        footnotes = [
            f"Daily table covers all {len(by)} dates with spend among indexed lines.",
            _list_footnote(len(txns), len(out_rows), header_n=_header_payments_made(job))
            or "Payment list includes every indexed outflow.",
        ]
        footnotes.extend(_coverage_footnotes(job))
        incomplete = _extraction_incomplete_blurb(job, len(txns))
        summary = (
            f"Daily spending across the statement — {n} payments totaling {_money(total, cur)}. "
            f"{len(by)} dates with spend are listed below."
        )
        if incomplete:
            summary = f"{incomplete} {summary}"
        return StructuredInsight(
            intent="date_lookup",
            intent_label=FINANCE_INTENT_LABELS["date_lookup"],
            headline="Spend by date",
            summary=summary,
            metrics=[
                InsightMetric(label="Total spent", value=_money(total, cur), tone="danger"),
                InsightMetric(label="Days with spend", value=str(len(by)), tone="neutral"),
                InsightMetric(label="Payments", value=str(n), tone="neutral"),
            ],
            table=table,
            transactions=txns,
            highlights=[],
            footnotes=footnotes,
        )

    day_month = _parse_day_month(question)
    if not day_month:
        month_only = _relative_month_key(question)
        if month_only:
            return build_month_spend(job, question, month_only)
        return build_spending_summary(job, question)

    day, month = day_month
    label = f"{day} {month.title()}"
    date_key = f"{day:02d} {month.title()[:3]}"
    by_text = _spend_by_date_from_text(_full_statement_text(job))
    bucket = by_text.get(date_key)
    rows = job.get("rows") or []
    matched_rows = [r for r in rows if _row_matches_day_month(r, day, month) and _money_out(r) > 0]
    text_hits = _day_hits_from_text(job, day, month)

    # Prefer complete statement-text blocks when available; fall back to table rows.
    if text_hits:
        matched = _dedupe_category_rows(text_hits + matched_rows)
        total = round(sum(_money_out(r) for r in matched), 2)
        count = len(matched)
    elif bucket:
        matched = matched_rows
        total = float(bucket["total"])
        count = int(bucket["n"])
    elif matched_rows:
        matched = matched_rows
        total = round(sum(_money_out(r) for r in matched), 2)
        count = len(matched)
    else:
        period = _period_label(job)
        count_q = _wants_payment_count(question)
        headline = f"Payments on {label}" if count_q else f"Spend on {label}"
        summary = (
            f"No payments on {label} on this statement."
            + (f" The statement covers {period}." if period else "")
            if count_q
            else (
                f"I searched “{filename}” for money-out activity on {label}. "
                "No matching payments showed up for that calendar day on this statement. "
                "You can try a nearby date or ask for spend by date."
            )
        )
        return StructuredInsight(
            intent="date_lookup",
            intent_label=FINANCE_INTENT_LABELS["date_lookup"],
            headline=headline,
            summary=summary,
            metrics=[
                InsightMetric(label="Date searched", value=label, tone="neutral"),
                InsightMetric(label="Result", value="No payments found", tone="accent"),
                *([InsightMetric(label="Statement period", value=period, tone="neutral")] if period else []),
            ],
            highlights=[],
            footnotes=[f"Looked through dated payment history for {label}."],
        )

    avg = total / max(count, 1)
    largest = max(matched, key=_money_out) if matched else None
    highlights = []
    if largest:
        highlights.append(
            f"Largest on this day: {_money(_money_out(largest), cur)} — {_clean_desc(str(largest.get('description') or ''))}"
        )
    if not _wants_payment_count(question):
        highlights.append(f"Average payment: {_money(avg, cur)}")

    txns = _format_txn_amounts(_txn_items(matched, out=True, full=True), cur) if matched else []
    footnotes = [f"Scoped to outflows dated {label} on “{filename}”."]
    if matched:
        note = _list_footnote(len(txns), len(matched))
        if note:
            footnotes.insert(0, note)
    if text_hits and len(text_hits) > len(matched_rows):
        footnotes.append(
            "Totals use the full statement text (table extract alone was incomplete for this day)."
        )

    count_q = _wants_payment_count(question)
    headline = f"Payments on {label}" if count_q else f"Spend on {label}"
    if count_q:
        summary = (
            f"{count} payment{'s' if count != 1 else ''} on {label}."
            + (f" All {len(txns)} indexed matches are listed below." if txns else "")
        )
    else:
        summary = (
            f"On {label}, {_money(total, cur)} went out across {count} payment{'s' if count != 1 else ''}."
            + (f" All {len(txns)} listed below." if txns else "")
        )

    return StructuredInsight(
        intent="date_lookup",
        intent_label=FINANCE_INTENT_LABELS["date_lookup"],
        headline=headline,
        summary=summary,
        metrics=[
            InsightMetric(label="Total spent", value=_money(total, cur), tone="danger"),
            InsightMetric(label="Payments", value=str(count), tone="neutral"),
            InsightMetric(label="Average", value=_money(avg, cur), tone="accent"),
            InsightMetric(label="Date", value=label, tone="neutral"),
        ],
        highlights=highlights,
        transactions=txns,
        footnotes=footnotes,
    )


def build_month_spend(job: dict[str, Any], question: str, month: str) -> StructuredInsight:
    """Answer spend scoped to a calendar month — never dump the whole statement."""
    cur = _cur(job)
    filename = job.get("filename") or "statement"
    mon = (month or "")[:3].lower()
    month_label = _MONTH_FULL.get(mon, mon.title())
    period = _period_label(job)

    rows = job.get("rows") or []
    matched_rows = [r for r in rows if _money_out(r) > 0 and _row_in_month(r, mon)]
    text_hits = _month_hits_from_text(job, mon)

    by_text = _spend_by_date_from_text(_full_statement_text(job))
    month_buckets = {
        k: v
        for k, v in by_text.items()
        if re.search(rf"\b{re.escape(mon)}\b", k.lower())
    }
    bucket_total = round(sum(float(v["total"]) for v in month_buckets.values()), 2) if month_buckets else 0.0
    bucket_n = sum(int(v["n"]) for v in month_buckets.values()) if month_buckets else 0

    if text_hits:
        matched = _dedupe_category_rows(text_hits + matched_rows)
        total = round(sum(_money_out(r) for r in matched), 2)
        count = len(matched)
    elif month_buckets and (not matched_rows or bucket_n >= len(matched_rows)):
        matched = matched_rows
        total = bucket_total
        count = bucket_n
    elif matched_rows:
        matched = matched_rows
        total = round(sum(_money_out(r) for r in matched), 2)
        count = len(matched)
    else:
        present = _months_present(job)
        present_labels = [_MONTH_FULL[m] for m in present]
        hint = (
            f" This statement mainly covers {', '.join(present_labels)}"
            + (f" ({period})" if period else "")
            + "."
            if present_labels
            else (f" Statement period: {period}." if period else "")
        )
        return StructuredInsight(
            intent="date_lookup",
            intent_label=FINANCE_INTENT_LABELS["date_lookup"],
            headline=f"Spend in {month_label}",
            summary=(
                f"I searched “{filename}” for money-out activity in {month_label}. "
                f"No matching payments showed up for that month on this statement.{hint} "
                "Try a month that appears on the PDF, or a specific date."
            ),
            metrics=[
                InsightMetric(label="Month", value=month_label, tone="neutral"),
                InsightMetric(label="Result", value="No payments found", tone="accent"),
                *([InsightMetric(label="Statement period", value=period, tone="neutral")] if period else []),
            ],
            highlights=[],
            footnotes=[f"Scoped to {month_label} only — not the full statement total."],
        )

    avg = total / max(count, 1)
    largest = max(matched, key=_money_out) if matched else None
    highlights = []
    if largest:
        highlights.append(
            f"Largest in {month_label}: {_money(_money_out(largest), cur)} — "
            f"{_clean_desc(str(largest.get('description') or ''))}"
        )
    if period:
        highlights.append(f"Statement covers {period} (answer filtered to {month_label}).")

    txns = _format_txn_amounts(_txn_items(matched, out=True, full=True), cur) if matched else []
    footnotes = [
        f"Scoped to outflows dated in {month_label} on “{filename}”.",
        "This is not the full statement spend — only the month you asked about.",
    ]
    if matched:
        note = _list_footnote(len(txns), len(matched))
        if note:
            footnotes.insert(0, note)

    table = None
    if month_buckets and len(month_buckets) > 1:
        table = InsightTable(
            headers=["Date", "Spent", "Payments"],
            rows=[
                [label, _money(month_buckets[label]["total"], cur) or "", str(int(month_buckets[label]["n"]))]
                for label in sorted(month_buckets.keys(), key=_date_sort_key)
            ],
        )

    return StructuredInsight(
        intent="date_lookup",
        intent_label=FINANCE_INTENT_LABELS["date_lookup"],
        headline=f"Spend in {month_label}",
        summary=(
            f"In {month_label}, {_money(total, cur)} went out across {count} payment{'s' if count != 1 else ''}."
            + (f" All {len(txns)} listed below." if txns else "")
        ),
        metrics=[
            InsightMetric(label="Total spent", value=_money(total, cur), tone="danger"),
            InsightMetric(label="Payments", value=str(count), tone="neutral"),
            InsightMetric(label="Average", value=_money(avg, cur), tone="accent"),
            InsightMetric(label="Month", value=month_label, tone="neutral"),
        ],
        highlights=highlights,
        transactions=txns,
        table=table,
        footnotes=footnotes,
    )


def _row_in_month(row: dict[str, Any], month: str) -> bool:
    blob = " ".join(
        str(row.get(k) or "") for k in ("date", "description", "raw", "account")
    ).lower()
    return bool(
        re.search(rf"\b{re.escape(month)}\b", blob)
        or f" {month}'" in blob
        or f"{month}'" in blob
    )


def _relative_month_key(question: str) -> str | None:
    """Resolve 'last month' / 'this month' / explicit month name to jan..dec."""
    q = (question or "").lower()
    today = datetime.now(timezone.utc).date()
    if re.search(r"\b(last|previous)\s+month\b", q):
        first = today.replace(day=1)
        prev = first - timedelta(days=1)
        return _MONTH_ORDER[prev.month - 1]
    if re.search(r"\b(this|current)\s+month\b", q):
        return _MONTH_ORDER[today.month - 1]
    return _month_token(question)


def _category_hits_from_text(
    job: dict[str, Any],
    keys: tuple[str, ...],
    *,
    month: str | None = None,
) -> list[dict[str, Any]]:
    """Parse Paytm-style payment blocks from full statement text.

    Tags like ``# Groceries`` often sit on a following line, so we group by
    date-headed blocks and match the whole block — not single lines.
    """

    def _block_matches(blob_low: str) -> bool:
        for key in keys:
            k = key.lower().strip()
            if not k:
                continue
            if k.startswith("#"):
                if k in blob_low:
                    return True
                continue
            if re.search(rf"(?<![a-z0-9]){re.escape(k)}(?![a-z0-9])", blob_low):
                return True
        return False

    hits: list[dict[str, Any]] = []
    for block in _payment_blocks(job):
        blob = "\n".join(block)
        low = blob.lower()
        if not _block_matches(low):
            continue
        row = _block_to_row(block)
        if not row:
            continue
        mon_key = _MONTH_ALIASES.get(row.get("_month") or "") or row.get("_month")
        if month and mon_key != month:
            continue
        hits.append(row)
    return hits


def _dedupe_category_rows(rows: list[dict[str, Any]]) -> list[dict[str, Any]]:
    """Prefer table-extracted rows; drop text duplicates of the same date+amount."""
    ordered = sorted(rows, key=lambda r: 1 if r.get("_from_text") else 0)
    seen: set[tuple] = set()
    out: list[dict[str, Any]] = []
    for r in ordered:
        date = (_row_date_label(r) or "").lower()
        date_key = re.sub(r"\s+", " ", date.split("\n")[0]).strip()
        date_key = re.sub(r"'?\d{2,4}$", "", date_key).strip()
        m = re.match(
            r"(?i)(\d{1,2})\s*(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)",
            date_key,
        )
        if m:
            date_key = f"{int(m.group(1)):02d} {m.group(2).lower()[:3]}"
        amt = round(_money_out(r), 2)
        key = (date_key, amt)
        if key in seen and amt > 0:
            continue
        seen.add(key)
        out.append(r)
    return out


def build_payment_count(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    """Answer how many payments were made/received from statement header totals."""
    m = _header_metrics(job)
    filename = job.get("filename") or "statement"
    period = _period_label(job)
    q = (question or "").lower()
    # Normalize light typos: ",ade" / "mad" → made
    q = re.sub(r"\b,\s*ade\b", " made", q)
    q = re.sub(r"\bmad\b", "made", q)

    made = m.get("payments_made")
    recv = m.get("payments_received")
    # Row fallback already in _header_metrics; keep local copies for clarity
    in_rows = _inflow_rows(job)
    out_rows = _outflow_rows(job)
    if made is None and out_rows:
        made = len(out_rows)
    if recv is None and in_rows:
        recv = len(in_rows)

    want_credit = bool(re.search(r"(?i)\b(credits?|deposits?|received|inflow)\b", q))
    want_debit = bool(re.search(r"(?i)\b(debits?|withdrawals?)\b", q))
    want_recv = (want_credit and not want_debit) or (
        bool(re.search(r"(?i)\breceiv", q)) and not re.search(r"(?i)\bmade\b", q)
    )
    want_made = (want_debit and not want_credit) or (
        (bool(re.search(r"(?i)\bmade\b", q)) or not want_recv) and not want_credit
    )
    if want_credit and not want_debit:
        want_recv = True
        want_made = False

    metrics: list[InsightMetric] = []
    if want_made and made is not None:
        metrics.append(InsightMetric(label="Payments made", value=str(int(made)), tone="danger"))
    if want_recv and recv is not None:
        label = "Credits / deposits" if want_credit else "Payments received"
        metrics.append(InsightMetric(label=label, value=str(int(recv)), tone="success"))
    if made is not None and recv is not None and want_made and not want_recv:
        metrics.append(InsightMetric(label="Payments received", value=str(int(recv)), tone="neutral"))
    if period:
        metrics.append(InsightMetric(label="Period", value=period, tone="neutral"))
    metrics.append(InsightMetric(label="Document", value=str(filename), tone="neutral"))

    if want_recv and not want_made and recv is not None:
        kind = "credit" if want_credit else "payment"
        summary = (
            f"This statement shows {int(recv)} {kind}{'s' if int(recv) != 1 else ''} "
            f"(money in) on “{filename}”."
        )
        headline = "Credits received" if want_credit else "Payments received"
    elif made is not None:
        extra = f" and {int(recv)} received" if recv is not None else ""
        summary = f"This statement shows {int(made)} payments made{extra}."
        headline = "Payment count"
    elif recv is not None:
        summary = f"This statement shows {int(recv)} payments received."
        headline = "Payment count"
    else:
        rows_n = len(job.get("rows") or [])
        summary = (
            f"I couldn’t find a payment count on “{filename}”. "
            f"I extracted {rows_n} transaction line{'s' if rows_n != 1 else ''} from the PDF."
        )
        headline = "Payment count"
        metrics = [InsightMetric(label="Extracted lines", value=str(rows_n), tone="neutral")]

    return StructuredInsight(
        intent="payment_count",
        intent_label=FINANCE_INTENT_LABELS.get("payment_count", "Payment Count"),
        headline=headline,
        summary=summary,
        metrics=metrics,
        highlights=[],
        footnotes=["Counts use statement header totals when available, otherwise indexed credit/debit lines."],
    )


def build_payment_list(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    """List outflow payments — lighter than a full spending summary."""
    cur = _cur(job)
    m = _header_metrics(job)
    # Same combined index as month/date lookups (text blocks + table rows).
    out_rows = _complete_outflow_rows(job)
    count = m.get("payments_made")
    header_n = _header_payments_made(job)
    period = _period_label(job)
    txns = _format_txn_amounts(_txn_items(out_rows, out=True, full=True), cur)
    listed = len(txns)

    if is_vague_single_payment_question(question or ""):
        header_show = int(count) if count is not None else listed
        metrics = [
            InsightMetric(label="Payments on statement", value=str(header_show), tone="neutral"),
        ]
        if period:
            metrics.append(InsightMetric(label="Period", value=period, tone="neutral"))
        return StructuredInsight(
            intent="transaction_search",
            intent_label=FINANCE_INTENT_LABELS["transaction_search"],
            headline="Which payment?",
            summary=(
                f"Your statement shows {header_show} payment{'s' if header_show != 1 else ''}. "
                "I can't tell which one you mean without more detail — try a merchant, date, or amount."
            ),
            metrics=metrics,
            footnotes=_coverage_footnotes(job),
        )

    metrics: list[InsightMetric] = []
    if header_n is not None:
        metrics.append(
            InsightMetric(label="Payments (header)", value=str(header_n), tone="neutral")
        )
    metrics.append(InsightMetric(label="Indexed & listed", value=str(listed), tone="accent"))
    if period:
        metrics.append(InsightMetric(label="Period", value=period, tone="neutral"))

    footnotes: list[str] = []
    note = _list_footnote(listed, len(out_rows), header_n=header_n)
    if note:
        footnotes.append(note)
    footnotes.extend(_coverage_footnotes(job))

    incomplete = _extraction_incomplete_blurb(job, listed)
    months = _months_present(job)
    month_hint = ""
    if months:
        labels = [_MONTH_FULL.get(m, m.title()) for m in months]
        month_hint = f" Indexed months include {', '.join(labels)}."

    if incomplete:
        summary = (
            f"{incomplete} Listing the {listed} indexed payment"
            f"{'s' if listed != 1 else ''} I do have below.{month_hint}"
        )
    elif listed:
        summary = (
            f"Here are all {listed} indexed payment{'s' if listed != 1 else ''} "
            f"I could retrieve from the statement.{month_hint}"
        )
    else:
        summary = "I couldn't find indexed payment lines on this statement."

    return StructuredInsight(
        intent="transaction_search",
        intent_label=FINANCE_INTENT_LABELS["transaction_search"],
        headline="Your payments",
        summary=summary,
        metrics=metrics,
        transactions=txns,
        footnotes=footnotes,
    )


def build_merchant_spending(job: dict[str, Any], question: str) -> StructuredInsight | None:
    """Sum spend for a named merchant/payee across rows + statement text."""
    name = extract_merchant_query(question)
    if not name:
        return None
    # If this is a known category alias, reuse category path (already narrowed).
    cat = _detect_spend_category(question)
    if cat and cat[0].lower() == name.lower():
        return build_category_spending(job, question)

    keys = (name.lower(),)
    # Also try first token for long names ("Nykaa ERetail" → still match "nykaa")
    tokens = [t for t in re.split(r"\s+", name.lower()) if len(t) > 2]
    if tokens and tokens[0] not in keys:
        keys = (name.lower(), tokens[0])

    cur = _cur(job)
    text_hits = _category_hits_from_text(job, keys)
    rows = [r for r in (job.get("rows") or []) if _money_out(r) > 0 and _row_matches_keys(r, keys)]
    matched = _dedupe_category_rows(text_hits + rows) if text_hits or rows else []

    display = name.title() if name.islower() else name
    if not matched:
        return StructuredInsight(
            intent="merchant_analysis",
            intent_label=FINANCE_INTENT_LABELS["merchant_analysis"],
            headline=f"{display} spend",
            summary=(
                f"I looked for payments to “{display}” on this statement and didn’t find clear matches. "
                "Try the exact merchant name as it appears on the PDF."
            ),
            metrics=[InsightMetric(label="Merchant", value=display, tone="neutral")],
            highlights=[],
            footnotes=[f"Searched narrations for: {', '.join(keys)}."],
        )

    txns = _format_txn_amounts(_txn_items(matched, out=True, full=True), cur)
    total = round(sum(_money_out(r) for r in matched), 2)
    count = len(txns)
    avg = total / max(count, 1)
    footnotes = [
        f"Showing all {count} matching payment{'s' if count != 1 else ''} to {display}.",
    ]
    if len(text_hits) > len(rows):
        footnotes.append(
            "Totals use the full statement text (table extract alone was incomplete)."
        )
    footnotes.extend(_coverage_footnotes(job))

    return StructuredInsight(
        intent="merchant_analysis",
        intent_label=FINANCE_INTENT_LABELS["merchant_analysis"],
        headline=f"{display} spend",
        summary=(
            f"Spending on {display}: {_money(total, cur)} across {count} payment{'s' if count != 1 else ''}. "
            "Every matching payment is listed below."
        ),
        metrics=[
            InsightMetric(label="Merchant total", value=_money(total, cur), tone="danger"),
            InsightMetric(label="Payments", value=str(count), tone="neutral"),
            InsightMetric(label="Average", value=_money(avg, cur), tone="accent"),
        ],
        transactions=txns,
        highlights=[
            f"{_money(_money_out(r), cur)} · {_clean_desc(str(r.get('description') or ''))}"
            for r in sorted(matched, key=_money_out, reverse=True)[:3]
        ],
        footnotes=footnotes,
    )


def build_category_spending(job: dict[str, Any], question: str) -> StructuredInsight | None:
    cat = _detect_spend_category(question)
    if not cat:
        return None
    label, keys = cat
    cur = _cur(job)
    month = _relative_month_key(question)
    month_label = _MONTH_FULL.get(month or "", "") if month else ""

    # Text blocks are the completeness source (table extract often misses lines).
    text_hits = _category_hits_from_text(job, keys, month=month)
    rows = [r for r in (job.get("rows") or []) if _money_out(r) > 0 and _row_matches_keys(r, keys)]
    if month:
        rows = [r for r in rows if _row_in_month(r, month)]

    if len(text_hits) >= len(rows):
        matched = _dedupe_category_rows(text_hits + rows)
    else:
        matched = _dedupe_category_rows(rows + text_hits)

    period_bit = f" in {month_label}" if month_label else ""
    if not matched:
        return StructuredInsight(
            intent="category_spending",
            intent_label=FINANCE_INTENT_LABELS["category_spending"],
            headline=f"{label.title()} spend",
            summary=(
                f"I looked for {label} activity{period_bit} on this statement and didn’t find clear matches. "
                "Try a merchant name that appears on the PDF, or ask for overall spending."
            ),
            metrics=[
                InsightMetric(label="Category", value=label.title(), tone="neutral"),
                *([InsightMetric(label="Period", value=month_label, tone="accent")] if month_label else []),
            ],
            highlights=[],
            footnotes=[f"Searched narrations for: {', '.join(keys[:6])}."],
        )

    txns = _format_txn_amounts(_txn_items(matched, out=True, full=True), cur)
    total = round(sum(_money_out(r) for r in matched), 2)
    count = len(txns)
    avg = total / max(count, 1)

    footnotes = [
        f"Showing all {count} matching {label} payment{'s' if count != 1 else ''}{period_bit}."
    ]
    if month_label:
        footnotes.append(
            f"Filtered to {month_label}. Payments outside this month on the statement are excluded."
        )
    if len(text_hits) > len(rows):
        footnotes.append(
            "Totals use the full statement text (table extract alone was incomplete)."
        )
    footnotes.extend(_coverage_footnotes(job))

    return StructuredInsight(
        intent="category_spending",
        intent_label=FINANCE_INTENT_LABELS["category_spending"],
        headline=f"{label.title()} spend{(' · ' + month_label) if month_label else ''}",
        summary=(
            f"Spending on {label}{period_bit}: {_money(total, cur)} across {count} payments. "
            f"{'Every matching indexed payment is listed below.' if count == len(txns) else 'Matching indexed payments are listed below.'}"
        ),
        metrics=[
            InsightMetric(label="Category total", value=_money(total, cur), tone="danger"),
            InsightMetric(label="Payments", value=str(count), tone="neutral"),
            InsightMetric(label="Average", value=_money(avg, cur), tone="accent"),
            *([InsightMetric(label="Period", value=month_label, tone="accent")] if month_label else []),
        ],
        transactions=txns,
        highlights=[
            f"{_money(_money_out(r), cur)} · {_clean_desc(str(r.get('description') or ''))}"
            for r in sorted(matched, key=_money_out, reverse=True)[:3]
        ],
        footnotes=footnotes,
    )



def build_coverage_explanation(job: dict[str, Any], question: str | None = None) -> StructuredInsight:
    """Answer follow-ups about payment list coverage — without exposing internals."""
    cur = _cur(job)
    m = _header_metrics(job)
    complete = _complete_outflow_rows(job)
    indexed = len(complete)
    header_n = _header_payments_made(job)
    period = _period_label(job)
    filename = job.get("filename") or "statement"

    metrics = [
        InsightMetric(label="Payments reviewed", value=str(indexed), tone="accent"),
    ]
    if header_n is not None:
        metrics.insert(
            0,
            InsightMetric(label="Payments on statement", value=str(header_n), tone="neutral"),
        )
    if period:
        metrics.append(InsightMetric(label="Period", value=period, tone="neutral"))
    if m.get("money_out") is not None:
        metrics.append(
            InsightMetric(label="Total spent", value=_money(m["money_out"], cur), tone="danger")
        )

    if header_n is not None and indexed < header_n:
        summary = (
            f"On “{filename}” I can review {indexed} individual payments in detail "
            f"(the statement reports {header_n} overall). "
            "Overall totals use the statement figures; ask about a date or merchant "
            "to inspect specific payments."
        )
        headline = "Payment coverage"
    else:
        summary = (
            f"I can review {indexed} payments on “{filename}”"
            + (f", matching the statement total of {header_n}." if header_n is not None else ".")
        )
        headline = "Payment coverage"

    return StructuredInsight(
        intent="period_coverage",
        intent_label="Coverage",
        headline=headline,
        summary=summary,
        metrics=metrics,
        highlights=[],
        footnotes=[],
    )


def _months_present(job: dict[str, Any]) -> list[str]:
    """Return ordered month keys (jan..dec) found on the statement."""
    found: set[str] = set()
    period = _period_label(job) or ""
    text = _full_statement_text(job)[:20000]
    by_text = _spend_by_date_from_text(text)
    by_rows = _spend_by_date_from_rows(job.get("rows") or [])

    for blob in (period, text[:4000]):
        for m in re.finditer(
            r"(?i)\b(jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|jun(?:e)?|"
            r"jul(?:y)?|aug(?:ust)?|sep(?:t(?:ember)?)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)\b",
            blob,
        ):
            key = _MONTH_ALIASES.get(m.group(1).lower())
            if key:
                found.add(key)

    for label in list(by_text.keys()) + list(by_rows.keys()):
        for m in re.finditer(
            r"(?i)\b(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\b",
            label,
        ):
            key = _MONTH_ALIASES.get(m.group(1).lower()[:3])
            if key:
                found.add(key)

    for r in job.get("rows") or []:
        date = _normalize_date_label(str(r.get("date") or "")) or str(r.get("date") or "")
        for m in re.finditer(
            r"(?i)\b(jan|feb|mar|apr|may|jun|jul|aug|sep|oct|nov|dec)[a-z]*\b",
            date,
        ):
            key = _MONTH_ALIASES.get(m.group(1).lower()[:3])
            if key:
                found.add(key)

    return [k for k in _MONTH_ORDER if k in found]


def _months_mentioned(question: str) -> list[str]:
    out: list[str] = []
    for m in re.finditer(
        r"(?i)\b(jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|jun(?:e)?|"
        r"jul(?:y)?|aug(?:ust)?|sep(?:t(?:ember)?)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)\b",
        question or "",
    ):
        key = _MONTH_ALIASES.get(m.group(1).lower())
        if key and key not in out:
            out.append(key)
    return out


def build_period_coverage(job: dict[str, Any], question: str) -> StructuredInsight:
    """Answer why certain months appear / don’t appear — from full statement, not chunks."""
    filename = job.get("filename") or "Your statement"
    period = _period_label(job)
    present = _months_present(job)
    present_labels = [_MONTH_FULL[m] for m in present]
    asked = _months_mentioned(question)
    missing = [m for m in asked if m not in present]
    present_asked = [m for m in asked if m in present]

    if period:
        summary = (
            f"“{filename}” covers **{period}**. "
            "That printed period is what the statement contains — not a retrieval sample."
        )
    elif present_labels:
        if len(present_labels) == 1:
            summary = (
                f"“{filename}” only includes activity in **{present_labels[0]}**. "
                "I checked the statement header and every dated transaction on the document."
            )
        else:
            joined = ", ".join(present_labels[:-1]) + f" and {present_labels[-1]}"
            summary = (
                f"“{filename}” includes activity in **{joined}**. "
                "I checked the statement header and every dated transaction on the document."
            )
    else:
        summary = (
            f"I reviewed “{filename}” but could not pin a clear calendar period from the header. "
            "Ask me to list spend by date for a day-by-day view."
        )

    highlights: list[str] = []
    if missing:
        miss_labels = [_MONTH_FULL[m] for m in missing]
        highlights.append(
            f"{', '.join(miss_labels)} "
            f"{'is' if len(miss_labels) == 1 else 'are'} not on this statement — "
            "there are no dated lines or header dates for "
            f"{'that month' if len(miss_labels) == 1 else 'those months'}."
        )
    if present_asked and not missing:
        highlights.append(
            "The month(s) you named do appear on the statement; ask about spend on a specific date for detail."
        )
    if present_labels and missing:
        highlights.append(
            f"Months actually present: {', '.join(present_labels)}."
        )

    metrics = []
    if period:
        metrics.append(InsightMetric(label="Statement period", value=period, tone="accent"))
    if present_labels:
        metrics.append(
            InsightMetric(
                label="Months present",
                value=", ".join(present_labels),
                tone="neutral",
            )
        )
    if missing:
        metrics.append(
            InsightMetric(
                label="Not on statement",
                value=", ".join(_MONTH_FULL[m] for m in missing),
                tone="danger",
            )
        )

    footnotes = _coverage_footnotes(job)
    footnotes.append("Answered from the full statement period and dated lines, not vector snippets.")

    return StructuredInsight(
        intent="period_coverage",
        intent_label=FINANCE_INTENT_LABELS["period_coverage"],
        headline="Statement period",
        summary=summary,
        metrics=metrics,
        highlights=highlights,
        footnotes=footnotes,
    )


def build_calc_explanation(
    job: dict[str, Any],
    *,
    prior_question: str | None = None,
) -> StructuredInsight:
    """Explain how totals were derived — prefer header, disclose row coverage."""
    cur = _cur(job)
    m = _header_metrics(job)
    out_rows = _outflow_rows(job)
    row_total = round(sum(_money_out(r) for r in out_rows), 2) if out_rows else None
    header_out = m.get("money_out")
    header_n = m.get("payments_made")
    period = _period_label(job)
    focus = (prior_question or "the previous figure").strip()

    highlights = [
        "Statement header totals are treated as primary for overall spend/receive.",
        f"Indexed outflow lines on this document: {len(out_rows)}"
        + (f" totaling {_money(row_total, cur)}" if row_total is not None else "")
        + ".",
    ]
    if header_out is not None:
        highlights.append(f"Header money out: {_money(header_out, cur)}.")
    if header_n is not None:
        highlights.append(f"Header payments made: {int(header_n)}.")

    footnotes = _coverage_footnotes(job)
    footnotes.append(f"Recomputed while explaining: {focus}")

    metrics = []
    if header_out is not None:
        metrics.append(InsightMetric(label="Header total spent", value=_money(header_out, cur), tone="danger"))
    if row_total is not None:
        metrics.append(InsightMetric(label="Sum of indexed outflows", value=_money(row_total, cur), tone="neutral"))
    if period:
        metrics.append(InsightMetric(label="Period", value=period, tone="accent"))

    return StructuredInsight(
        intent="insights",
        intent_label="Calculation check",
        headline="How the figures were derived",
        summary=(
            "I re-checked the full statement. Overall cash totals come from the statement header when present; "
            "date/merchant/category cuts aggregate every matching indexed transaction on the document."
        ),
        metrics=metrics,
        highlights=highlights,
        footnotes=footnotes,
    )


def build_kb_inventory(org_id: str, agent_id: str, question: str | None = None) -> StructuredInsight | None:
    """Answer knowledge-base / document-metadata questions — never summarize PDF contents."""
    all_docs = saas_db.list_kb_documents(org_id, agent_id) or []
    ready = unique_ready_docs([d for d in all_docs if d.get("status") == "ready"])
    pending = [d for d in all_docs if d.get("status") in {"pending", "processing", "indexing"}]
    failed = [d for d in all_docs if d.get("status") in {"failed", "error"}]

    q = (question or "").strip()
    wants_pages = bool(re.search(r"(?i)\bpages?\b", q))
    wants_filename = bool(re.search(r"(?i)\bfilename|file\s+name|document\s+name\b", q))
    wants_count = bool(re.search(r"(?i)\bhow\s+many\s+(?:documents?|files?|pdfs?)\b", q))
    wants_status = bool(re.search(r"(?i)\bstatus|ready|uploaded|indexed\b", q))

    if not all_docs:
        return StructuredInsight(
            intent="document_qa",
            intent_label="Knowledge base",
            headline="Knowledge base is empty",
            summary=(
                "I don't have a knowledge base yet, so I can't list filenames or page counts. "
                "Please upload a knowledge base first (PDF, TXT, MD, or CSV in the Knowledge panel) — "
                "I won’t invent files or page counts."
            ),
            metrics=[
                InsightMetric(label="Documents", value="0", tone="neutral"),
                InsightMetric(label="Ready", value="0", tone="neutral"),
            ],
            highlights=[],
            footnotes=["Metadata only — no document content was summarized."],
        )

    if not ready and (pending or failed):
        summary = (
            f"I see {len(all_docs)} upload{'s' if len(all_docs) != 1 else ''}, "
            f"but none are ready yet"
            + (f" ({len(pending)} still processing)" if pending else "")
            + (f", {len(failed)} failed" if failed else "")
            + ". I won’t guess page counts or contents until indexing finishes."
        )
        return StructuredInsight(
            intent="document_qa",
            intent_label="Knowledge base",
            headline="Uploads in progress",
            summary=summary,
            metrics=[
                InsightMetric(label="Total uploads", value=str(len(all_docs)), tone="neutral"),
                InsightMetric(label="Ready", value="0", tone="danger"),
                InsightMetric(label="Processing", value=str(len(pending)), tone="accent"),
            ],
            highlights=[str(d.get("filename") or "file") for d in (pending + failed)[:6]],
            footnotes=["Ask again once status is ready."],
        )

    names = [str(d.get("filename") or "document") for d in ready]
    total_pages = sum(int(d.get("page_count") or 0) for d in ready)
    total_chunks = sum(int(d.get("chunk_count") or 0) for d in ready)

    metrics: list[InsightMetric] = [
        InsightMetric(label="Ready documents", value=str(len(ready)), tone="accent"),
    ]
    if total_pages:
        metrics.append(InsightMetric(label="Total pages", value=str(total_pages), tone="neutral"))
    if total_chunks:
        metrics.append(InsightMetric(label="Indexed chunks", value=str(total_chunks), tone="neutral"))
    if pending:
        metrics.append(InsightMetric(label="Still processing", value=str(len(pending)), tone="neutral"))

    table = InsightTable(
        headers=["Filename", "Status", "Pages", "Chunks"],
        rows=[
            [
                str(d.get("filename") or "document"),
                str(d.get("status") or "ready"),
                str(int(d.get("page_count") or 0) or "—"),
                str(int(d.get("chunk_count") or 0) or "—"),
            ]
            for d in ready[:12]
        ],
    )

    if wants_filename and len(ready) == 1:
        summary = f"The indexed file is “{names[0]}” (status: ready)."
        if total_pages:
            summary += f" It has {total_pages} page{'s' if total_pages != 1 else ''} indexed."
    elif wants_pages:
        if total_pages:
            summary = (
                f"Across {len(ready)} ready document{'s' if len(ready) != 1 else ''}, "
                f"I have {total_pages} page{'s' if total_pages != 1 else ''} indexed."
            )
        else:
            summary = (
                f"I have {len(ready)} ready document{'s' if len(ready) != 1 else ''}, "
                "but page counts were not recorded for these uploads — I won’t invent a page number."
            )
    elif wants_count:
        summary = f"You have {len(ready)} ready document{'s' if len(ready) != 1 else ''} in the knowledge base."
    elif wants_status:
        summary = (
            f"{len(ready)} document{'s' if len(ready) != 1 else ''} ready"
            + (f", {len(pending)} still processing" if pending else "")
            + (f", {len(failed)} failed" if failed else "")
            + "."
        )
    else:
        listed = ", ".join(f"“{n}”" for n in names[:5])
        extra = f" and {len(names) - 5} more" if len(names) > 5 else ""
        summary = (
            f"Your knowledge base has {len(ready)} ready document{'s' if len(ready) != 1 else ''}: "
            f"{listed}{extra}."
        )
        if total_pages:
            summary += f" Combined indexed pages: {total_pages}."
        summary += " Ask about spend or a merchant if you want the statement contents — not just the file list."

    return StructuredInsight(
        intent="document_qa",
        intent_label="Knowledge base",
        headline="Indexed documents",
        summary=summary,
        metrics=metrics,
        highlights=[f"“{n}”" for n in names[:8]],
        table=table,
        footnotes=[
            "This is inventory metadata (filenames, pages, status) — not a content summary of the PDFs."
        ],
    )


def try_structured_kb_answer(
    org_id: str,
    agent_id: str,
    question: str,
    *,
    original_question: str | None = None,
    conv_intent: str | None = None,
    prior_question: str | None = None,
    prior_answer: str | None = None,
) -> tuple[str, StructuredInsight] | None:
    """Return analyst markdown + structured insight when we can answer from statements.

    Always aggregates over the full statement job (header + all indexed rows/text),
    never over a vector-chunk sample.
    """
    q = (question or "").strip()
    original = (original_question or q).strip()
    if not q and not original:
        return None

    # Classify on the user's wording first. Only fall back to the history-expanded
    # rewrite when the original is already analytical or an explicit follow-up
    # (never let "how can you help" inherit spend from a prior turn).
    intent = classify_finance_intent(original)
    if intent == "general" and q != original and conv_intent in {"follow_up", "verify", "explain_calc"}:
        rewritten_intent = classify_finance_intent(q)
        if rewritten_intent != "general":
            intent = rewritten_intent

    # Pure KB inventory asks — which statement / knowledge base files
    from app.saas.knowledge_state import is_kb_meta_question as _is_kb_meta

    kb_meta = _is_kb_meta(original) or _is_kb_meta(q) or bool(
        re.search(r"(?i)\bknowledge\s*base\b", original + " " + q)
    )
    overview_ask = is_analyse_overview_ask(original) or is_analyse_overview_ask(q)
    if kb_meta and not overview_ask:
        inv = build_kb_inventory(org_id, agent_id, original)
        if inv:
            return insight_to_markdown(inv), inv

    jobs = list_agent_statement_jobs(org_id, agent_id)
    if not jobs:
        if re.search(r"(?i)\bknowledge\s*base\b", original + " " + q):
            docs = saas_db.list_kb_documents(org_id, agent_id)
            ready = [d for d in docs if d.get("status") == "ready"]
            if not ready:
                empty = StructuredInsight(
                    intent="document_qa",
                    intent_label="Document Q&A",
                    headline="Knowledge base is empty",
                    summary="Upload a PDF, TXT, MD, or CSV to teach this agent, then ask again.",
                    metrics=[],
                    highlights=[],
                    footnotes=[],
                )
                return insight_to_markdown(empty), empty
        return None

    job = jobs[0]

    # Challenges / calc explain: re-check data, do not defend prior prose.
    if conv_intent == "explain_calc" or is_coverage_challenge(original) or is_coverage_challenge(q):
        if is_coverage_challenge(original) or is_coverage_challenge(q):
            insight = build_coverage_explanation(job, original)
            return insight_to_markdown(insight), insight
        insight = build_calc_explanation(job, prior_question=prior_question or original)
        return insight_to_markdown(insight), insight

    if conv_intent == "verify":
        # Re-answer the prior analytical question from full data when possible.
        focus = prior_question or original
        focus_intent = classify_finance_intent(focus)
        if is_period_coverage_question(original) or is_period_coverage_question(q):
            insight = build_period_coverage(job, original)
        elif focus_intent == "date_lookup" or is_per_date_spend_question(focus):
            insight = build_date_lookup(job, focus)
        elif focus_intent == "category_spending":
            insight = build_category_spending(job, focus) or build_spending_summary(job, focus)
        elif focus_intent == "income":
            insight = build_income_summary(job, focus)
        elif focus_intent in {"spending_summary", "insights", "comparison", "merchant_analysis"}:
            insight = build_spending_summary(job, focus)
        elif focus_intent == "period_coverage" or is_period_coverage_question(focus):
            insight = build_period_coverage(job, focus)
        else:
            insight = build_statement_summary(job, focus)
        insight.footnotes = list(insight.footnotes or []) + [
            "Re-checked against the full statement after your challenge — not defending the earlier reply."
        ]
        if prior_answer:
            insight.highlights = list(insight.highlights or [])
            insight.highlights.insert(
                0,
                "Prior answer was set aside; figures below are freshly aggregated from the document.",
            )
        return insight_to_markdown(insight), insight

    # Period / month coverage (e.g. "why only July not August") — before overview.
    if (
        intent == "period_coverage"
        or is_period_coverage_question(original)
        or is_period_coverage_question(q)
    ):
        insight = build_period_coverage(job, original if is_period_coverage_question(original) else q)
        return insight_to_markdown(insight), insight

    if is_profit_loss_question(original) or is_profit_loss_question(q):
        insight = build_cashflow_verdict(job, original)
        return insight_to_markdown(insight), insight

    if intent == "financial_advice" or is_financial_advice_question(original) or is_financial_advice_question(q):
        insight = build_financial_analysis(job, original)
        return insight_to_markdown(insight), insight

    if intent == "date_lookup" or is_per_date_spend_question(original):
        insight = build_date_lookup(job, original)
        extra = _coverage_footnotes(job)
        if extra:
            insight.footnotes = list(insight.footnotes or []) + extra
        return insight_to_markdown(insight), insight

    # Only use rewritten wording for date lookup when the original itself is vague/follow-up.
    if (
        conv_intent in {"follow_up", "verify"}
        and is_per_date_spend_question(q)
        and intent in {"general", "document_qa"}
    ):
        insight = build_date_lookup(job, q)
        extra = _coverage_footnotes(job)
        if extra:
            insight.footnotes = list(insight.footnotes or []) + extra
        return insight_to_markdown(insight), insight

    if intent == "payment_count":
        insight = build_payment_count(job, original)
        return insight_to_markdown(insight), insight

    paragraph_ask = wants_paragraph_format(original) or wants_paragraph_format(q)
    payment_context = (
        is_payment_list_question(original)
        or (_wants_full_list(original) and re.search(r"(?i)\bpayments?\b", original))
        or (
            paragraph_ask
            and prior_question
            and (
                is_payment_list_question(prior_question)
                or re.search(r"(?i)\b(?:list|payments?|transactions?)\b", prior_question)
            )
        )
    )
    if payment_context and (
        is_payment_list_question(original)
        or (_wants_full_list(original) and re.search(r"(?i)\bpayments?\b", original))
        or paragraph_ask
    ):
        focus = original
        if paragraph_ask and prior_question and conv_intent in {"follow_up", "verify"}:
            focus = prior_question
        insight = build_payment_list(job, focus)
        if paragraph_ask:
            insight = apply_paragraph_style(insight, original)
        return insight_to_markdown(insight), insight

    if is_payment_list_question(original) or (
        _wants_full_list(original) and re.search(r"(?i)\bpayments?\b", original)
    ):
        insight = build_payment_list(job, original)
        return insight_to_markdown(insight), insight

    if intent == "category_spending":
        insight = build_category_spending(job, original)
        if insight:
            insight.footnotes = list(insight.footnotes or []) + _coverage_footnotes(
                job, matched_n=len(insight.transactions or [])
            )
            return insight_to_markdown(insight), insight

    if intent == "income":
        insight = build_income_summary(job, original)
        insight.footnotes = list(insight.footnotes or []) + _coverage_footnotes(job)
        return insight_to_markdown(insight), insight

    if intent in {"spending_summary", "insights", "comparison"}:
        if is_profit_loss_question(original):
            insight = build_cashflow_verdict(job, original)
            return insight_to_markdown(insight), insight
        if is_biggest_expenses_question(original) or (
            intent == "insights" and re.search(r"(?i)\b(biggest|largest|top\s+\d+)\b", original)
        ):
            insight = build_biggest_expenses(job, original)
            return insight_to_markdown(insight), insight
        # Named merchant inside a spend ask → focused merchant cut, not whole statement.
        merchant_insight = build_merchant_spending(job, original)
        if merchant_insight and extract_merchant_query(original):
            return insight_to_markdown(merchant_insight), merchant_insight
        insight = build_spending_summary(job, original)
        if intent == "insights":
            insight.intent = "insights"
            insight.intent_label = FINANCE_INTENT_LABELS["insights"]
            insight.headline = "Spending insights"
        return insight_to_markdown(insight), insight

    if is_pdf_overview_question(original) or overview_ask:
        if not re.search(r"(?i)\bknowledge\s*base\b", original):
            insight = build_statement_summary(job, original)
            return insight_to_markdown(insight), insight

    if intent == "statement_summary":
        insight = build_statement_summary(job, original)
        return insight_to_markdown(insight), insight

    # Follow-ups that still look "general" but refer to the loaded statement
    if conv_intent in {"follow_up", "verify"} and prior_question:
        if is_coverage_challenge(original) or is_coverage_challenge(q):
            insight = build_coverage_explanation(job, original)
            return insight_to_markdown(insight), insight
        prior_intent = classify_finance_intent(prior_question)
        if prior_intent == "period_coverage" or is_period_coverage_question(original):
            insight = build_period_coverage(job, original)
            return insight_to_markdown(insight), insight
        if wants_paragraph_format(original) and prior_question and re.search(
            r"(?i)\b(?:list|payments?|transactions?)\b", prior_question
        ):
            insight = build_payment_list(job, prior_question)
            insight = apply_paragraph_style(insight, original)
            return insight_to_markdown(insight), insight
        if _wants_full_list(original):
            if re.search(r"(?i)\bpayments?\b", original) or (
                prior_question and re.search(r"(?i)\bpayments?\b", prior_question)
            ):
                insight = build_payment_list(job, prior_question or original)
                if wants_paragraph_format(original):
                    insight = apply_paragraph_style(insight, original)
                return insight_to_markdown(insight), insight
        # Re-answer prior analytical question with follow-up context (no generic fallback).
        if prior_intent == "date_lookup" or is_per_date_spend_question(prior_question):
            insight = build_date_lookup(job, prior_question)
            insight.summary = (
                f"Following up on “{prior_question}”: {insight.summary}"
            )
            return insight_to_markdown(insight), insight
        if prior_intent in {"spending_summary", "insights", "merchant_analysis"}:
            if is_biggest_expenses_question(prior_question) or is_biggest_expenses_question(original):
                insight = build_biggest_expenses(job, prior_question or original)
                return insight_to_markdown(insight), insight
            insight = build_spending_summary(job, prior_question)
            insight.summary = f"Following up on your earlier spend question: {insight.summary}"
            return insight_to_markdown(insight), insight
        if prior_intent == "transaction_search" or is_payment_list_question(prior_question):
            insight = build_payment_list(job, prior_question)
            return insight_to_markdown(insight), insight
        if prior_intent == "income":
            insight = build_income_summary(job, prior_question)
            return insight_to_markdown(insight), insight
        if prior_intent == "category_spending":
            insight = build_category_spending(job, prior_question) or build_spending_summary(
                job, prior_question
            )
            return insight_to_markdown(insight), insight
        if prior_intent in {
            "statement_summary",
            "spending_summary",
            "date_lookup",
            "income",
            "insights",
            "payment_count",
            "transaction_search",
        }:
            # Prefer answering the follow-up about coverage / period when months appear
            if _months_mentioned(original):
                insight = build_period_coverage(job, original)
                return insight_to_markdown(insight), insight
            if is_coverage_challenge(original) or re.search(
                r"(?i)\b(missing|incomplete|only\s+\d+|rest\s+of)\b", original
            ):
                insight = build_coverage_explanation(job, original)
                return insight_to_markdown(insight), insight
            # Stay on the prior analytical track instead of a generic fallback.
            if prior_intent == "statement_summary":
                insight = build_statement_summary(job, prior_question)
            elif prior_intent == "income":
                insight = build_income_summary(job, prior_question)
            elif prior_intent == "payment_count":
                insight = build_payment_count(job, prior_question)
            elif prior_intent == "transaction_search":
                insight = build_payment_list(job, prior_question)
            elif prior_intent == "date_lookup":
                insight = build_date_lookup(job, prior_question)
            else:
                insight = build_spending_summary(job, prior_question)
            insight.summary = f"Following up: {insight.summary}"
            return insight_to_markdown(insight), insight

    if intent in {"merchant_analysis", "transaction_search"}:
        if intent == "transaction_search" and is_payment_list_question(original):
            insight = build_payment_list(job, original)
            return insight_to_markdown(insight), insight
        merchant_insight = build_merchant_spending(job, original)
        if merchant_insight:
            return insight_to_markdown(merchant_insight), merchant_insight
        insight = build_spending_summary(job, original)
        insight.intent = intent
        insight.intent_label = FINANCE_INTENT_LABELS.get(intent, "Analysis")
        insight.footnotes = list(insight.footnotes or []) + [
            "Name a merchant (e.g. Zepto or a person) for a focused payment list.",
        ]
        return insight_to_markdown(insight), insight

    return None