"""Unit tests for statement accuracy fixes (merchant, count, date, third-party)."""

from __future__ import annotations

import re

from app.pdf.calc import _detect_spend_category
from app.saas.answer_style import apply_paragraph_style, transactions_to_paragraph, wants_paragraph_format
from app.saas.chat_router import route_chat
from app.saas.finance_intent import (
    classify_finance_intent,
    extract_merchant_query,
    is_analyse_overview_ask,
    is_payment_count_question,
    is_payment_list_question,
    is_unrelated_party_spend_question,
    is_vague_single_payment_question,
)
from app.saas.insight import (
    _account_holder,
    build_date_lookup,
    build_merchant_spending,
    build_payment_count,
    build_payment_list,
    build_category_spending,
    build_month_spend,
    build_spending_summary,
    build_statement_summary,
)


def test_zepto_is_merchant_not_full_groceries():
    cat = _detect_spend_category("How much spent on Zepto?")
    assert cat is not None
    label, keys = cat
    assert label == "zepto"
    assert keys == ("zepto",)
    assert "blinkit" not in keys


def test_groceries_still_expands():
    cat = _detect_spend_category("How much spent on groceries?")
    assert cat is not None
    label, keys = cat
    assert label == "groceries"
    assert "zepto" in keys and "blinkit" in keys


def test_payment_count_intent():
    assert is_payment_count_question("How many payments were made?")
    assert classify_finance_intent("How many payments were made?") == "payment_count"


def test_third_party_spend_refused():
    q = "How much did Elon Musk spend last month?"
    assert is_unrelated_party_spend_question(q)
    assert classify_finance_intent(q) == "general"
    r = route_chat(q)
    assert r.target == "template_offtopic"
    assert not r.use_docs


def test_first_person_spend_still_docs():
    q = "How much did I spend last month?"
    assert not is_unrelated_party_spend_question(q)
    r = route_chat(q)
    assert r.use_docs


def test_extract_merchant_nykaa():
    assert extract_merchant_query("How much did I pay Nykaa?")
    name = extract_merchant_query("How much did I pay Nykaa?")
    assert name and "nykaa" in name.lower()


def test_statement_identity_and_period_intent():
    assert classify_finance_intent("Whose statement is this?") == "statement_summary"
    assert classify_finance_intent("What is the statement period?") == "statement_summary"


def _sample_job() -> dict:
    return {
        "filename": "sample.pdf",
        "statement_totals": {
            "money_out": 1000.0,
            "money_in": 100.0,
            "payments_made": 84,
            "payments_received": 5,
            "source": "statement_header",
        },
        "calculation": {"metrics": {}},
        "rows": [
            {
                "date": "16 Jul",
                "description": "Paid to Nykaa ERetail Private Limited",
                "debit": 993.0,
                "credit": None,
                "amount": -993.0,
            },
            {
                "date": "16 Jul",
                "description": "Money sent to Aditya Yadav",
                "debit": 3000.0,
                "credit": None,
                "amount": -3000.0,
            },
            {
                "date": "06 Jul",
                "description": "Paid to Zepto",
                "debit": 327.0,
                "credit": None,
                "amount": -327.0,
            },
            {
                "date": "11 Jul",
                "description": "Paid to Blinkit",
                "debit": 177.0,
                "credit": None,
                "amount": -177.0,
            },
        ],
        "document_text": (
            "AAYUSHI VERMA\n"
            "Paytm Statement\n"
            "3 JUL'26 - 2 AUG'26 - Rs.30,298.30\n"
            "84 Payments made 5 Payments received\n"
            "16 Jul Paid to Nykaa ERetail Private Limited Note: HDFC Bank - - Rs.993\n"
            "10:54 PM\n"
            "16 Jul Money sent to Jatin Thakur Tag: State Bank - Rs.89\n"
            "7:13 PM\n"
            "16 Jul Money sent to Aditya Yadav Tag: State Bank - Rs.3,000\n"
            "4:04 PM\n"
            "16 Jul Money sent to Aditya Yadav Tag: HDFC Bank - - Rs.1\n"
            "4:03 PM\n"
            "16 Jul Money sent to Bhairab Pathak Tag: HDFC Bank - - Rs.114\n"
            "9:03 AM\n"
            "06 Jul Paid to Zepto Note: UPIIntent HDFC Bank - - Rs.327\n"
            "3:46 PM\n"
            "11 Jul Paid to Zepto Note: UPIIntent HDFC Bank - - Rs.326\n"
            "12:12 PM\n"
            "11 Jul Paid to Blinkit Note: HDFC Bank - - Rs.177\n"
            "1:00 PM\n"
        ),
    }


def test_payment_count_uses_header():
    insight = build_payment_count(_sample_job(), "How many payments were made?")
    assert "84" in insight.summary
    assert any(m.value == "84" for m in insight.metrics)


def test_date_lookup_uses_full_text_day():
    insight = build_date_lookup(_sample_job(), "How much did I spend on 16 July?")
    # 993+89+3000+1+114 = 4197
    assert any("4,197" in (m.value or "") or "4197" in (m.value or "").replace(",", "") for m in insight.metrics)
    assert len(insight.transactions or []) == 5


def test_zepto_category_not_blinkit():
    insight = build_category_spending(_sample_job(), "How much spent on Zepto?")
    assert insight is not None
    descs = " ".join(t.description for t in (insight.transactions or [])).lower()
    assert "zepto" in descs
    assert "blinkit" not in descs


def test_account_holder_from_header():
    holder = _account_holder(_sample_job())
    assert holder and "aayushi" in holder.lower()


def test_unique_ready_docs_dedupes_filename():
    from app.saas.knowledge_state import unique_ready_docs

    docs = [
        {"status": "ready", "filename": "Paytm.pdf", "created_at": "1"},
        {"status": "ready", "filename": "Paytm.pdf", "created_at": "2"},
        {"status": "ready", "filename": "Other.csv", "created_at": "1"},
    ]
    out = unique_ready_docs(docs)
    assert len(out) == 2
    assert {d["filename"] for d in out} == {"Paytm.pdf", "Other.csv"}


def test_merchant_nykaa():
    insight = build_merchant_spending(_sample_job(), "How much did I pay Nykaa?")
    assert insight is not None
    assert any("993" in (m.value or "").replace(",", "") for m in insight.metrics)


def test_hdfc_branch_and_credits_from_statement():
    from pathlib import Path

    from app.saas.kb_answer import build_statement_job
    from app.saas.insight import build_payment_count, build_statement_field_answer
    from app.saas.finance_intent import classify_finance_intent

    path = Path("/home/esfera/Documents/accountants_chatbot/Acct Statement_4640_14062026_10.58.32.pdf")
    if not path.exists():
        return
    job = build_statement_job(path, path.name)
    assert job
    assert classify_finance_intent("what is the state account branch number?") == "statement_summary"
    field = build_statement_field_answer(job, "what is the state account branch number?")
    vals = " ".join(m.value for m in field.metrics)
    assert "1395" in vals  # branch code
    assert "HDFC0001395" in vals or "SECTOR" in vals.upper()

    assert classify_finance_intent("how many credits were ,ade?") == "payment_count"
    credits = build_payment_count(job, "how many credits were ,ade?")
    assert credits.headline.lower().startswith("credit")
    assert any(m.label.lower().startswith("credit") and int(m.value) >= 1 for m in credits.metrics)


def test_which_payment_routes_to_transaction_search():
    assert is_payment_list_question("which payment did i make")
    assert classify_finance_intent("which payment did i make") == "transaction_search"
    assert is_vague_single_payment_question("which payment did i make")
    assert not is_vague_single_payment_question("which payments did i make")


def test_which_payment_singular_clarifies():
    insight = build_payment_list(_sample_job(), "which payment did i make")
    assert insight.headline == "Which payment?"
    assert "can't tell which one" in insight.summary.lower()
    assert not insight.table
    assert not insight.transactions


def test_which_payments_plural_lists_without_categories():
    insight = build_payment_list(_sample_job(), "which payments did i make")
    assert insight.headline == "Your payments"
    assert insight.transactions
    assert not insight.table
    assert "Spending summary" not in insight.headline


def test_month_spend_intent_not_full_summary():
    assert classify_finance_intent("how much did i spend in august?") == "date_lookup"
    assert classify_finance_intent("how much did I spend in July?") == "date_lookup"


def test_august_spend_scoped_not_full_dump():
    job = _sample_job()
    # Sample rows are all July — August should not dump the whole statement.
    insight = build_month_spend(job, "how much did i spend in august?", "aug")
    assert insight.headline == "Spend in August"
    assert "Spending summary" not in insight.headline
    assert "august" in insight.summary.lower() or "August" in insight.summary
    # No July payment list dumped when month has no matches
    assert not insight.transactions
    assert any("No payments" in (m.value or "") for m in insight.metrics)


def test_july_spend_scoped():
    insight = build_date_lookup(_sample_job(), "how much did i spend in july?")
    assert insight.headline == "Spend in July"
    assert insight.transactions
    assert all("jul" in (t.date or "").lower() for t in insight.transactions if t.date)


def test_refined_routing_intents():
    assert is_analyse_overview_ask("analyse the docment")
    assert not is_analyse_overview_ask("how much did i spend in total?")
    assert not is_analyse_overview_ask("list all payments")
    assert classify_finance_intent("how much did i spend in total?") == "spending_summary"
    assert classify_finance_intent("list all payments") == "transaction_search"
    assert classify_finance_intent("how many payments did i make in 1 august") == "date_lookup"


def test_compact_overview_no_payment_list():
    insight = build_statement_summary(_sample_job(), "analyse the document")
    assert insight.headline == "Statement overview"
    assert not insight.transactions
    assert any(m.label == "Total spent" for m in insight.metrics)
    # Narrative should observe, not dump "Overall: total received … total spent …"
    assert "overall:" not in insight.summary.lower()
    assert "net" in insight.summary.lower() or "outflow" in insight.summary.lower() or "inflow" in insight.summary.lower()


def test_total_spend_compact_not_overview():
    insight = build_spending_summary(_sample_job(), "how much did i spend in total?")
    assert insight.headline == "Spending summary"
    assert not insight.transactions
    assert insight.summary.lower().startswith("you spent")
    assert "largest" in insight.summary.lower()


def test_list_payments_light_not_overview():
    insight = build_payment_list(_sample_job(), "list all payments")
    assert insight.headline == "Your payments"
    assert insight.transactions
    assert "Statement overview" not in insight.headline


def test_biggest_expenses_ranked_list():
    from app.saas.finance_intent import is_biggest_expenses_question
    from app.saas.insight import build_biggest_expenses

    q = "Show my biggest expenses"
    assert is_biggest_expenses_question(q)
    assert classify_finance_intent(q) == "insights"
    insight = build_biggest_expenses(_sample_job(), q)
    assert insight.headline == "Biggest expenses"
    assert insight.transactions
    assert insight.table and insight.table.headers[:3] == ["#", "Date", "Merchant / description"]
    # Ranked largest-first
    amounts = []
    for t in insight.transactions:
        raw = (t.amount or "").replace(",", "").replace("₹", "").replace("Rs.", "").strip()
        try:
            amounts.append(float(re.sub(r"[^\d.]", "", raw) or 0))
        except ValueError:
            pass
    assert amounts == sorted(amounts, reverse=True)
    # Should not dump the full totals overview as the headline story
    assert "overall:" not in insight.summary.lower()


def test_general_knowledge_skips_rag():
    for q in (
        "what is the capital of France",
        "who invented the telephone",
        "what is photosynthesis",
        "tell me a joke",
    ):
        r = route_chat(q)
        assert r.use_docs is False, (q, r)
        assert r.target == "ollama_concept", (q, r)


def test_payment_list_never_claims_all_when_incomplete():
    job = _sample_job()
    # Header says 84; sample only has a handful of rows.
    insight = build_payment_list(job, "list all payments")
    assert insight.transactions
    text = (insight.summary + " " + " ".join(insight.footnotes or [])).lower()
    assert "incomplete" in text or "indexed" in text
    assert not re.search(r"showing all 84", text)
    assert "showing all" not in text or "indexed" in text


def test_payment_list_includes_text_august_with_month():
    """Full payment list must use the same text+rows index as month lookups."""
    job = _sample_job()
    # Inject an August payment only in document text (not in table rows).
    job = {
        **job,
        "document_text": job["document_text"]
        + "\n01 Aug Money sent to Test Payee Tag: HDFC Bank - - Rs.250\n10:00 AM\n",
        "rows": list(job["rows"]),  # July-only table rows
    }
    month = build_month_spend(job, "how much did i spend in august?", "aug")
    assert month.transactions, "August text hit should surface in month spend"
    assert any("aug" in (t.date or "").lower() for t in month.transactions)

    listing = build_payment_list(job, "list all payments")
    assert any("aug" in (t.date or "").lower() for t in listing.transactions), (
        "August text payments must appear in list-all, same as month query"
    )


def test_coverage_challenge_explains_gap():
    from app.saas.conversation import is_coverage_challenge
    from app.saas.insight import build_coverage_explanation

    assert is_coverage_challenge("why only 40 payments?")
    assert is_coverage_challenge("where are the rest of the transactions")
    insight = build_coverage_explanation(_sample_job(), "why only 4 of 84?")
    assert "payments" in insight.summary.lower()
    assert "84" in insight.summary or any("84" in (m.value or "") for m in insight.metrics)
    assert insight.headline == "Payment coverage"
    assert "extraction" not in insight.summary.lower()
    assert "indexed" not in insight.summary.lower()


def test_list_footnote_honest():
    from app.saas.insight import _list_footnote

    note = _list_footnote(40, 40, header_n=84)
    assert note is not None
    assert "40" in note and "84" in note
    assert "extraction" not in note.lower()
    complete = _list_footnote(84, 84, header_n=84)
    assert complete is not None
    assert "all 84" in complete.lower()


def test_august_payment_count_date_lookup():
    insight = build_date_lookup(_sample_job(), "how many payments did i make in 1 august")
    assert insight.headline.startswith("Payments on 1 Aug")
    assert "no payments" in insight.summary.lower()
    assert not insight.transactions


def test_vague_what_routes_to_clarify():
    assert route_chat("what").target == "template_clarify"
    assert route_chat("gi").target == "template_greeting"


def test_full_list_paragraph_intent():
    assert classify_finance_intent("i need full list of payments in a paragraph") == "transaction_search"
    assert wants_paragraph_format("in a paragraph")


def test_payment_list_paragraph_style():
    insight = build_payment_list(_sample_job(), "list all payments in a paragraph")
    styled = apply_paragraph_style(insight, "list all payments in a paragraph")
    assert not styled.transactions
    assert "paid" in styled.summary.lower()
    assert len(styled.summary) > 80
