"""RAG chatbot service — retrieval + Ollama reasoning."""

from __future__ import annotations

import difflib
import json
import re
import time
from collections import defaultdict, deque
from collections.abc import Iterator
from functools import lru_cache

import httpx
from langchain_community.embeddings import HuggingFaceEmbeddings
from langchain_community.vectorstores import Chroma

from app.config import Settings, get_settings
from app.models.schemas import ChatResponse, Citation
from app.rag.prompts import (
    BYE_PROMPT,
    CONCEPT_PROMPT,
    GREETING_PROMPT,
    HELP_PROMPT,
    NO_DATA_PROMPT,
    OFFTOPIC_PROMPT,
    PDF_SYSTEM_PROMPT,
    SYSTEM_PROMPT,
    THANKS_PROMPT,
    USER_PROMPT,
)
from app.rag.calculator import calculate as calc_expression
from app.rag.calculator import is_pure_math
from app.rag.structured import run_structured_query
from app.pdf.calc import format_job_context
from app.pdf.calc import answer_statement_metric_question, is_pdf_overview_question, describe_pdf_job
from app.pdf.store import get_session_job

# ── Intent helpers ────────────────────────────────────────────────────────────

_GREETING_WORDS = {
    "hi", "hii", "hhi", "hello", "hey", "hey there", "good morning",
    "good afternoon", "good evening", "howdy", "yo", "sup", "what's up", "whats up",
    "wassup", "wassupp", "whassup", "whatsup", "what's good", "whats good",
    # incomplete / typo greetings (e.g. accidental "h")
    "h", "he", "hel", "hell", "hai", "helo", "hllo", "hellp", "hola",
    "kello", "kelloo", "heloo", "heelo", "hullo", "hallo",
    # wellbeing smalltalk
    "how are you", "how are you doing", "how r you", "how r u", "how are u",
    "how you doing", "how's it going", "hows it going", "how is it going",
    "how have you been", "you good", "you ok",
}
_THANKS_WORDS = {"thanks", "thank you", "thx", "thank u"}
_BYE_WORDS = {"bye", "goodbye", "see you", "see ya"}
_HELP_WORDS = {
    "who are you",
    "who r you",
    "what can you do",
    "what can you help me with",
    "what can you help with",
    "how can you help",
    "how can you help me",
    "tell me about yourself",
    "tell me about you",
    "tell me about the bot",
    "tell me about this bot",
    "tell me about the assistant",
    "about you",
    "about yourself",
    "help",
    "help me",
}
_HELP_RE = re.compile(
    r"(?i)\b("
    r"how\s+can\s+you\s+help(?:\s+me)?|"
    r"what\s+can\s+you\s+(?:do|help(?:\s+me)?(?:\s+with)?)|"
    r"what\s+do\s+you\s+know|"
    r"who\s+(?:are|r)\s+you|"
    r"tell\s+me\s+about\s+(?:yourself|you|the\s+bot|this\s+bot|the\s+assistant|alex)|"
    r"(?:about\s+yourself|about\s+you)\b|"
    r"help\s+me\b|\bhelp\b"
    r")\b"
)

_HOW_ARE_YOU_RE = re.compile(
    r"(?i)^("
    r"how\s+are\s+(?:you|u)(?:\s+doing)?|"
    r"how(?:'s|\s+is)\s+it\s+going|"
    r"how\s+have\s+you\s+been|"
    r"how\s+you\s+doing|"
    r"you\s+(?:good|ok|okay)|"
    r"(?:what'?s|whats)\s+up|"
    r"wass?u+p+|whass?u+p+|"
    r"(?:what'?s|whats)\s+good|"
    r"yo|sup"
    r")[?.!\s]*$"
)

# General / casual topics — answer with the base LLM (not document RAG).
_OFFTOPIC_RE = re.compile(
    r"\b(weather|football|cricket|movie|song|joke|recipe|"
    r"(?:write|debug|run)\s+code|python\s+code|javascript\s+code|"
    r"programming|bitcoin price today|horoscope|dating|girlfriend|boyfriend|"
    r"capital\s+of|who\s+invented|who\s+discovered|photosynthesis|gravity|"
    r"solar\s+system|planet|continent|ocean|mountain|president\s+of|"
    r"population\s+of|distance\s+between|how\s+tall\s+is|how\s+old\s+is\s+(?:the|earth)|"
    r"tell\s+me\s+a\s+joke|fun\s+fact"
    r")\b",
    re.IGNORECASE,
)

# Conceptual questions — answer from knowledge, skip noisy retrieval
_CONCEPT_RE = re.compile(
    r"^\s*(what\s+is|what\s+are|what\'s|define|definition\s+of|explain|meaning\s+of|"
    r"how\s+does|how\s+do|difference\s+between|why\s+is|why\s+are|"
    r"who\s+(?:is|was|invented|discovered|wrote|painted)|"
    r"where\s+is|when\s+was|when\s+did|which\s+(?:country|city|planet)|"
    r"tell\s+me\s+about(?!\s+(?:my|the\s+statement|the\s+pdf|the\s+document|this\s+pdf))"
    r")\b",
    re.IGNORECASE,
)

_LEAK_PATTERNS = [
    re.compile(r"(?im)^\s*sources?\s*:.*$"),
    re.compile(r"(?im)^\s*note\s+\d+\s*:.*$"),
    re.compile(r"(?i)\b(record_id|txn_\d+|source\s*\d+|dataset|knowledge base|RAG|vector|structured results?|line item totals|reference notes)\b"),
    re.compile(r"(?im)^\s*(as an ai|based on (the )?(context|notes|provided)).*$"),
]

_GOODBYE_RE = re.compile(
    r"(?i)\b(goodbye|good bye|bye for now|take care|see you|farewell)\b[^.!?]*[.!?]?",
)

# Forbidden canned PDF replies
_CANNED_PDF_REPLY_RE = re.compile(
    r'(?is)^[\"“]?[^\"”\n]+\.pdf[\"”]?\s+is a balance sheet with \d+ line items\.'
    r'\s*Key balances:.*Ratios:',
)
_LOADED_ACK_RE = re.compile(
    r'(?is)^Loaded\s+[\"“].+[\"”]\.?\s*(Ask what|You can ask).+',
)


def _normalize_smalltalk(text: str) -> str:
    s = text.strip().lower()
    # Strip trailing punctuation / typos like "hi]" "hello!!!"
    s = re.sub(r"[^\w\s]+$", "", s).strip()
    s = re.sub(r"^alex[,.]?\s+", "", s)
    s = re.sub(r"\s+alex$", "", s).strip()
    s = re.sub(r"[^\w\s]+$", "", s).strip()
    return s


def _likely_greeting(norm: str) -> bool:
    """Typo-tolerant hello/hi/hey detection (e.g. KELLO → hello)."""
    if not norm:
        return False
    if norm in _GREETING_WORDS or norm == "alex":
        return True
    if len(norm) > 10 or not norm.replace(" ", "").isalpha():
        return False
    return bool(
        difflib.get_close_matches(norm, ["hello", "hi", "hey", "howdy", "hola"], n=1, cutoff=0.72)
    )


def _detect_intent(question: str) -> str:
    """Return: greeting | thanks | bye | help | offtopic | concept | data"""
    text = question.strip()
    if not text:
        return "greeting"

    norm = _normalize_smalltalk(text)

    if norm in _BYE_WORDS:
        return "bye"
    if norm in _THANKS_WORDS:
        return "thanks"
    # "help me analyse the pdf" is analysis, not capability help
    if (
        (norm in _HELP_WORDS or _HELP_RE.search(text.split(",")[0].strip()) or _HELP_RE.search(text))
        and not re.search(r"(?i)\b(analy[sz]e|summar(?:y|ize)|overview)\b", text)
        and not re.search(r"(?i)\b(spend|spent|payment|merchant|statement|pdf|document)\b", text)
    ):
        return "help"
    if norm in _GREETING_WORDS or norm == "alex" or _HOW_ARE_YOU_RE.match(norm) or _likely_greeting(norm):
        return "greeting"
    # Accidental single-key / very short non-numeric pings
    if len(norm) <= 2 and norm.isalpha() and not re.search(r"\d", text):
        return "greeting"
    if re.match(
        r"^(hi+|hii|hhi|hello+|hey+|good morning|good afternoon|good evening|howdy|yo)"
        r"(\s+(alex|there))?$",
        norm,
    ):
        return "greeting"
    if _OFFTOPIC_RE.search(text):
        return "offtopic"
    if is_pure_math(text):
        return "math"
    # Don't treat wellbeing smalltalk as a "concept" / data question
    if _HOW_ARE_YOU_RE.match(norm):
        return "greeting"
    if _CONCEPT_RE.search(text) and len(text.split()) <= 20:
        return "concept"
    return "data"


def _clean_answer(text: str, *, allow_goodbye: bool = False) -> str:
    """Strip model artifacts and enforce a single coherent reply."""
    cleaned = text.strip()
    cleaned = re.sub(r"^(Alex|Assistant|Reply)\s*:\s*", "", cleaned, flags=re.IGNORECASE)

    lines = []
    for line in cleaned.splitlines():
        if any(p.search(line) for p in _LEAK_PATTERNS[:2]):
            continue
        lines.append(line)
    cleaned = "\n".join(lines).strip()
    cleaned = _LEAK_PATTERNS[2].sub("", cleaned)

    if not allow_goodbye:
        cleaned = _GOODBYE_RE.sub("", cleaned)

    # Keep only the first short paragraph if the model dumps multiple intents
    parts = re.split(r"\n\s*\n", cleaned)
    cleaned = parts[0].strip() if parts else cleaned

    # Cap to ~3 sentences for conversational clarity
    sentences = re.split(r"(?<=[.!?])\s+", cleaned)
    if len(sentences) > 3:
        cleaned = " ".join(sentences[:3]).strip()

    cleaned = re.sub(r"\s{2,}", " ", cleaned).strip()
    cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).strip()
    return cleaned or "I don't have enough information to answer that confidently."


# Short per-session memory for follow-up questions (in-process only)
_SESSION_HISTORY: dict[str, deque[tuple[str, str]]] = defaultdict(lambda: deque(maxlen=3))


def clear_session_history(session_id: str | None) -> None:
    if not session_id:
        return
    _SESSION_HISTORY.pop(session_id, None)

_INTENT_PROMPTS = {
    "greeting": GREETING_PROMPT,
    "thanks": THANKS_PROMPT,
    "bye": BYE_PROMPT,
    "help": HELP_PROMPT,
    "offtopic": OFFTOPIC_PROMPT,
    "concept": CONCEPT_PROMPT,
}


class RAGService:
    def __init__(self, settings: Settings):
        self.settings = settings
        self._embeddings = HuggingFaceEmbeddings(
            model_name=settings.embedding_model,
            model_kwargs={"device": "cpu"},
            encode_kwargs={"normalize_embeddings": True},
        )
        self._vectorstore: Chroma | None = None

    @property
    def embeddings(self) -> HuggingFaceEmbeddings:
        return self._embeddings

    def get_vectorstore(self) -> Chroma:
        if self._vectorstore is None:
            self.settings.chroma_persist_dir.mkdir(parents=True, exist_ok=True)
            self._vectorstore = Chroma(
                collection_name=self.settings.chroma_collection,
                embedding_function=self._embeddings,
                persist_directory=str(self.settings.chroma_persist_dir),
            )
        return self._vectorstore

    def get_llm(self) -> Ollama:
        if self._llm is None:
            llm_kwargs: dict = {
                "base_url": self.settings.ollama_base_url,
                "model": self.settings.ollama_model,
                "temperature": 0.0,
                "num_ctx": self.settings.ollama_num_ctx,
                "num_predict": self.settings.ollama_num_predict,
                # Keep model loaded between turns (big win on CPU).
                "keep_alive": "30m",
            }
            threads = self.settings.ollama_num_threads or os.cpu_count()
            if threads:
                llm_kwargs["num_thread"] = threads
            self._llm = Ollama(**llm_kwargs)
        return self._llm

    def warmup(self) -> None:
        """Pre-load embeddings, vector store, and keep Ollama model warm."""
        self.document_count()
        try:
            self._invoke_ollama("hi", num_predict=8)
        except Exception:
            pass

    def document_count(self) -> int | None:
        try:
            store = self.get_vectorstore()
            return store._collection.count()
        except Exception:
            return None

    def vector_store_ready(self) -> bool:
        count = self.document_count()
        return count is not None and count > 0

    async def ollama_reachable(self) -> bool:
        try:
            async with httpx.AsyncClient(timeout=5.0) as client:
                response = await client.get(f"{self.settings.ollama_base_url}/api/tags")
                return response.status_code == 200
        except Exception:
            return False

    def _format_context(self, docs) -> tuple[str, list[Citation]]:
        blocks: list[str] = []
        citations: list[Citation] = []

        for idx, doc in enumerate(docs, start=1):
            meta = doc.metadata
            record_id = meta.get("record_id", f"chunk_{idx}")
            category = meta.get("data_category", "unknown")
            source = meta.get("source_dataset", "unknown")
            excerpt = doc.page_content[:400].strip()
            max_chars = self.settings.max_context_chars_per_doc
            content = doc.page_content[:max_chars].strip()
            if len(doc.page_content) > max_chars:
                content += "..."

            blocks.append(f"Note {idx}:\n{content}")
            citations.append(
                Citation(
                    record_id=record_id,
                    data_category=category,
                    source_dataset=source,
                    excerpt=excerpt,
                )
            )

        return "\n\n---\n\n".join(blocks), citations

    def _history_block(self, session_id: str | None) -> str:
        if not session_id:
            return ""
        turns = _SESSION_HISTORY.get(session_id)
        if not turns:
            return ""
        lines = ["Recent conversation:"]
        for q, a in turns:
            lines.append(f"User: {q}")
            lines.append(f"Alex: {a[:180]}")
        return "\n".join(lines) + "\n\n"

    def _remember(self, session_id: str | None, question: str, answer: str) -> None:
        if not session_id:
            return
        _SESSION_HISTORY[session_id].append((question, answer))

    def _retrieve(self, question: str, session_id: str | None = None):
        """Retrieve relevant docs, optionally biased by last user question."""
        store = self.get_vectorstore()
        query = question
        if session_id and _SESSION_HISTORY.get(session_id):
            last_q, _ = _SESSION_HISTORY[session_id][-1]
            if len(question.split()) <= 8:
                query = f"{last_q}\n{question}"

        scored = store.similarity_search_with_relevance_scores(
            query, k=self.settings.retrieval_top_k
        )
        threshold = self.settings.similarity_threshold
        return [doc for doc, score in scored if score is not None and score >= threshold]

    def _ollama_predict(self, num_predict: int | None) -> tuple[int, bool, float, str]:
        think = self.settings.ollama_think
        predict = num_predict if num_predict is not None else self.settings.ollama_num_predict
        if think:
            predict = max(predict, self.settings.ollama_num_predict_thinking)
        else:
            predict = max(predict, self.settings.ollama_num_predict_min)
        base = self.settings.ollama_base_url.rstrip("/")
        timeout = self.settings.ollama_timeout * (2.0 if think else 1.0)
        return predict, think, timeout, base

    def _ollama_chat_payload(
        self, prompt: str, *, num_predict: int, use_think: bool
    ) -> dict:
        return {
            "model": self.settings.ollama_model,
            "messages": [{"role": "user", "content": prompt}],
            "stream": False,
            "think": use_think,
            "keep_alive": "30m",
            "options": {
                "temperature": 0.0,
                "num_ctx": self.settings.ollama_num_ctx,
                "num_predict": num_predict,
            },
        }

    def stream_ollama(
        self, prompt: str, *, num_predict: int | None = None
    ) -> Iterator[str]:
        """Stream visible answer tokens from Ollama /api/chat."""
        predict, think, timeout, base = self._ollama_predict(num_predict)
        payload = self._ollama_chat_payload(prompt, num_predict=predict, use_think=think)
        payload["stream"] = True

        try:
            with httpx.Client(timeout=timeout) as client:
                with client.stream("POST", f"{base}/api/chat", json=payload) as response:
                    if response.status_code == 404:
                        model_name = self.settings.ollama_model
                        raise RuntimeError(
                            f"Ollama model '{model_name}' was not found. "
                            f"Pull it with: ollama pull {model_name}"
                        )
                    if response.status_code != 200:
                        raise RuntimeError(
                            f"Ollama returned HTTP {response.status_code}: "
                            f"{response.read().decode()[:500]}"
                        )
                    for line in response.iter_lines():
                        if not line:
                            continue
                        try:
                            data = json.loads(line)
                        except json.JSONDecodeError:
                            continue
                        if err := data.get("error"):
                            raise RuntimeError(f"Ollama error: {err}")
                        chunk = (data.get("message") or {}).get("content") or ""
                        if chunk:
                            yield chunk
        except httpx.TimeoutException as exc:
            raise RuntimeError(
                f"Ollama timed out after {timeout}s at {base}"
            ) from exc
        except RuntimeError:
            raise
        except Exception as exc:
            raise RuntimeError(f"Ollama request failed: {exc}") from exc

    def _invoke_ollama(self, prompt: str, *, num_predict: int | None = None) -> str:
        """Call Ollama /api/chat (same API as curl) on the configured GPU/local server."""
        predict, think, timeout, base = self._ollama_predict(num_predict)

        def _chat(num_predict: int, *, use_think: bool) -> tuple[dict, str]:
            payload = self._ollama_chat_payload(
                prompt, num_predict=num_predict, use_think=use_think
            )
            try:
                with httpx.Client(timeout=timeout) as client:
                    response = client.post(f"{base}/api/chat", json=payload)
            except httpx.TimeoutException as exc:
                raise RuntimeError(
                    f"Ollama timed out after {timeout}s at {base}"
                ) from exc
            except Exception as exc:
                raise RuntimeError(f"Ollama request failed: {exc}") from exc

            if response.status_code == 404:
                model_name = self.settings.ollama_model
                raise RuntimeError(
                    f"Ollama model '{model_name}' was not found. "
                    f"Pull it with: ollama pull {model_name}"
                )
            if response.status_code != 200:
                raise RuntimeError(
                    f"Ollama returned HTTP {response.status_code}: {response.text[:500]}"
                )

            data = response.json()
            if err := data.get("error"):
                raise RuntimeError(f"Ollama error: {err}")
            content = (data.get("message") or {}).get("content") or ""
            return data, content

        data, content = _chat(predict, use_think=think)

        if not content.strip():
            if think:
                retry_predict = max(predict * 2, self.settings.ollama_num_predict_thinking * 2)
                data, content = _chat(retry_predict, use_think=True)
            else:
                data, content = _chat(max(predict, 512), use_think=False)

        if not content.strip() and data.get("done_reason") == "length":
            raise RuntimeError(
                f"Ollama returned an empty reply (num_predict={predict} may be too low for "
                f"{self.settings.ollama_model}; try raising OLLAMA_NUM_PREDICT_THINKING)"
            )
        return content.strip()

    def _intent_answer(self, question: str, intent: str) -> str:
        """Ollama reply for a single detected intent — no chat history (avoids mix-ups)."""
        template = _INTENT_PROMPTS.get(intent, NO_DATA_PROMPT)
        raw = self._invoke_ollama(
            template.format(question=question),
            num_predict=self.settings.ollama_num_predict_conversational,
        )
        return _clean_answer(raw, allow_goodbye=(intent == "bye"))

    def _no_data_answer(self, question: str) -> str:
        raw = self._invoke_ollama(
            NO_DATA_PROMPT.format(question=question),
            num_predict=self.settings.ollama_num_predict_conversational,
        )
        return _clean_answer(raw, allow_goodbye=False)

    def _finish_response(
        self,
        question: str,
        answer_text: str,
        citations: list[Citation],
        t0: float,
        session_id: str | None,
        mode: str,
        *,
        allow_goodbye: bool = False,
    ) -> ChatResponse:
        from app import database

        if mode in ("calculator", "pdf_overview"):
            answer_text = (answer_text or "").strip()
        else:
            answer_text = _clean_answer(answer_text, allow_goodbye=allow_goodbye)
        duration_ms = int((time.monotonic() - t0) * 1000)
        citations_json = json.dumps([c.model_dump() for c in citations])
        log_id = database.log_query(question, answer_text, citations_json, duration_ms, session_id)

        self._remember(session_id, question, answer_text)

        return ChatResponse(
            answer=answer_text,
            citations=citations,
            session_id=session_id,
            log_id=log_id,
            mode=mode,
            pdf_job_id=None,
            pdf_summary=None,
        )

    def chat(self, question: str, session_id: str | None = None) -> ChatResponse:
        t0 = time.monotonic()
        question = question.strip()
        intent = _detect_intent(question)

        # Calculator math — exact result (works with or without a PDF)
        if intent == "math":
            answer_text = calc_expression(question)
            return self._finish_response(
                question,
                answer_text,
                [],
                t0,
                session_id,
                "calculator",
            )

        # While a PDF is bound to this session, answer from that statement.
        pdf_job = get_session_job(session_id)
        has_pdf = bool(
            pdf_job
            and (
                pdf_job.get("rows")
                or pdf_job.get("calculation")
                or pdf_job.get("document_text")
                or pdf_job.get("text_preview")
                or pdf_job.get("statement_totals")
            )
        )

        if has_pdf:
            if intent == "bye":
                answer_text = self._intent_answer(question, intent)
                return self._finish_response(
                    question,
                    answer_text,
                    [],
                    t0,
                    session_id,
                    "conversational",
                    allow_goodbye=True,
                )

            # Exact totals for spend/received/net — don't let the model invent figures
            direct = answer_statement_metric_question(pdf_job, question)
            if direct:
                resp = self._finish_response(
                    question, direct, [], t0, session_id, "pdf_overview"
                )
                resp.pdf_job_id = pdf_job.get("job_id")
                resp.pdf_summary = (pdf_job.get("calculation") or {}).get("summary")
                return resp

            if is_pdf_overview_question(question):
                answer_text = describe_pdf_job(pdf_job)
                resp = self._finish_response(
                    question, answer_text, [], t0, session_id, "pdf_overview"
                )
                resp.pdf_job_id = pdf_job.get("job_id")
                resp.pdf_summary = (pdf_job.get("calculation") or {}).get("summary")
                return resp

            pdf_context = format_job_context(pdf_job)
            history = self._history_block(session_id)
            prompt = (
                PDF_SYSTEM_PROMPT.format(context=pdf_context)
                + "\n\n"
                + history
                + USER_PROMPT.format(question=question)
            )
            answer_text = self._invoke_ollama(
                prompt,
                num_predict=max(280, self.settings.ollama_num_predict),
            )
            if _CANNED_PDF_REPLY_RE.search((answer_text or "").strip()) or _LOADED_ACK_RE.search(
                (answer_text or "").strip()
            ):
                retry_prompt = (
                    prompt
                    + "\n\nRewrite your answer in natural sentences from the PDF text. "
                    "Do not say Loaded…, do not start with the filename, "
                    "and do not use 'Key balances' / 'Ratios:' labels."
                )
                answer_text = self._invoke_ollama(
                    retry_prompt,
                    num_predict=max(280, self.settings.ollama_num_predict),
                )
            resp = self._finish_response(question, answer_text, [], t0, session_id, "pdf")
            resp.pdf_job_id = pdf_job.get("job_id")
            resp.pdf_summary = (pdf_job.get("calculation") or {}).get("summary")
            return resp

        # No PDF — normal conversational / RAG routing
        if intent in _INTENT_PROMPTS:
            answer_text = self._intent_answer(question, intent)
            return self._finish_response(
                question,
                answer_text,
                [],
                t0,
                session_id,
                "conversational",
                allow_goodbye=(intent == "bye"),
            )

        # Data questions — prefer structured pandas filters when applicable
        structured = run_structured_query(question)
        relevant = self._retrieve(question, session_id)

        if structured:
            context_parts = [f"Structured results (authoritative):\n{structured}"]
            citations: list[Citation] = []
            if relevant:
                rag_context, citations = self._format_context(relevant)
                context_parts.append(f"Related notes:\n{rag_context}")
            context = "\n\n".join(context_parts)
            history = self._history_block(session_id)
            prompt = (
                SYSTEM_PROMPT.format(context=context)
                + "\n\n"
                + history
                + USER_PROMPT.format(question=question)
            )
            answer_text = self._invoke_ollama(prompt)
            return self._finish_response(question, answer_text, citations, t0, session_id, "rag")

        if not relevant:
            answer_text = self._no_data_answer(question)
            return self._finish_response(question, answer_text, [], t0, session_id, "conversational")

        context, citations = self._format_context(relevant)
        history = self._history_block(session_id)
        prompt = (
            SYSTEM_PROMPT.format(context=context)
            + "\n\n"
            + history
            + USER_PROMPT.format(question=question)
        )
        answer_text = self._invoke_ollama(prompt)
        return self._finish_response(question, answer_text, citations, t0, session_id, "rag")


@lru_cache
def get_rag_service() -> RAGService:
    return RAGService(get_settings())
