"""Local free OCR provider using Tesseract + OCRmyPDF (no GPU, no paid APIs)."""

from __future__ import annotations

import logging
import shutil
import tempfile
from pathlib import Path

from app.services.ocr.base import OCRResult
from app.services.ocr.invoice_extractor import extract_invoice_fields

logger = logging.getLogger(__name__)

# Minimum characters to treat embedded PDF text as usable (skip OCR).
_MIN_TEXT_CHARS = 40


def _tesseract_available() -> bool:
    if shutil.which("tesseract") is None:
        return False
    try:
        import pytesseract

        pytesseract.get_tesseract_version()
        return True
    except Exception:
        return False


def _ocrmypdf_available() -> bool:
    if shutil.which("ocrmypdf") is None and shutil.which("gs") is None:
        # ocrmypdf CLI may be a python entrypoint; still try import
        pass
    try:
        import ocrmypdf  # noqa: F401

        return True
    except Exception:
        return False


def _pdf2image_available() -> bool:
    if shutil.which("pdftoppm") is None:
        return False
    try:
        import pdf2image  # noqa: F401

        return True
    except Exception:
        return False


def _extract_pdf_text(path: Path) -> tuple[str, int]:
    """Extract embedded text from a PDF. Returns (text, page_count)."""
    text_parts: list[str] = []
    pages = 0
    try:
        from pypdf import PdfReader

        reader = PdfReader(str(path))
        pages = len(reader.pages)
        for page in reader.pages:
            try:
                part = page.extract_text() or ""
            except Exception:
                part = ""
            if part.strip():
                text_parts.append(part)
    except Exception as exc:
        logger.info("pypdf text extract failed for %s: %s", path.name, exc)

    joined = "\n".join(text_parts).strip()
    if len(joined) >= _MIN_TEXT_CHARS:
        return joined, pages or 1

    # Fallback: pdfminer (sometimes better on odd encodings)
    try:
        from pdfminer.high_level import extract_text

        mined = (extract_text(str(path)) or "").strip()
        if len(mined) > len(joined):
            joined = mined
    except Exception as exc:
        logger.info("pdfminer extract failed for %s: %s", path.name, exc)

    return joined, pages or (1 if joined else 0)


def _preprocess_image(img):
    """Lightweight preprocessing — does not mutate the file on disk."""
    from PIL import Image, ImageOps, ImageFilter

    # Convert to RGB then grayscale
    if img.mode not in ("L", "RGB"):
        img = img.convert("RGB")
    gray = ImageOps.grayscale(img)
    # Upscale small images for better OCR
    w, h = gray.size
    if max(w, h) < 1200:
        scale = 1200 / max(w, h)
        gray = gray.resize((int(w * scale), int(h * scale)))
    # Mild contrast + sharpen
    gray = ImageOps.autocontrast(gray)
    gray = gray.filter(ImageFilter.SHARPEN)
    return gray


def _ocr_image_file(path: Path) -> str:
    import pytesseract
    from PIL import Image

    with Image.open(path) as img:
        processed = _preprocess_image(img)
        # Prefer Dutch+English when available; fall back to eng
        for lang in ("nld+eng", "eng"):
            try:
                return pytesseract.image_to_string(processed, lang=lang) or ""
            except Exception:
                continue
        return pytesseract.image_to_string(processed) or ""


def _ocr_pdf_via_ocrmypdf(path: Path) -> tuple[str, int]:
    """Run OCRmyPDF into a temp file (original untouched), then extract text."""
    import ocrmypdf

    with tempfile.TemporaryDirectory(prefix="amanah-ocr-") as tmp:
        out = Path(tmp) / "ocr.pdf"
        # force-ocr only when needed; here caller already decided text is insufficient
        ocrmypdf.ocr(
            str(path),
            str(out),
            force_ocr=True,
            language=["eng", "nld"],
            deskew=True,
            rotate_pages=True,
            progress_bar=False,
            optimize=0,
        )
        return _extract_pdf_text(out)


def _ocr_pdf_via_images(path: Path) -> tuple[str, int]:
    """Fallback: rasterize PDF pages with poppler and OCR each page."""
    import pytesseract
    from pdf2image import convert_from_path

    images = convert_from_path(str(path), dpi=200)
    parts: list[str] = []
    for img in images:
        processed = _preprocess_image(img)
        try:
            text = pytesseract.image_to_string(processed, lang="nld+eng")
        except Exception:
            text = pytesseract.image_to_string(processed, lang="eng")
        if text and text.strip():
            parts.append(text)
        img.close()
    return "\n\n".join(parts), len(images)


class LocalTesseractOCRProvider:
    """Free local OCR — Tesseract for images, OCRmyPDF/pdf2image for scanned PDFs."""

    name = "tesseract"

    def available(self) -> bool:
        return _tesseract_available()

    def process_document(self, file_path: str, mime_type: str | None) -> OCRResult:
        path = Path(file_path)
        if not path.is_file():
            return OCRResult(
                text="",
                pages=0,
                ocr_engine=self.name,
                status="failed",
                error="Uploaded file could not be found for OCR.",
            )

        if not self.available():
            return OCRResult(
                text="",
                pages=0,
                ocr_engine=self.name,
                status="unavailable",
                error="Local OCR engine is not installed or unavailable.",
            )

        mime = (mime_type or "").lower().strip()
        suffix = path.suffix.lower()

        try:
            if mime == "application/pdf" or suffix == ".pdf":
                return self._process_pdf(path)
            if mime.startswith("image/") or suffix in {".jpg", ".jpeg", ".png", ".webp"}:
                return self._process_image(path)
            return OCRResult(
                text="",
                pages=0,
                ocr_engine=self.name,
                status="failed",
                error="Unsupported file type for OCR.",
            )
        except Exception as exc:
            logger.exception("OCR failed for %s", path.name)
            return OCRResult(
                text="",
                pages=0,
                ocr_engine=self.name,
                status="failed",
                error="Local OCR could not read this document.",
                meta={"detail": str(exc)[:200]},
            )

    def _finalize(self, text: str, pages: int, *, used_ocr: bool, engine: str) -> OCRResult:
        cleaned = (text or "").strip()
        extraction = extract_invoice_fields(cleaned)
        return OCRResult(
            text=cleaned,
            pages=max(pages, 1 if cleaned else 0),
            ocr_engine=engine,
            status="completed",
            used_ocr=used_ocr,
            extraction=extraction,
        )

    def _process_image(self, path: Path) -> OCRResult:
        text = _ocr_image_file(path)
        return self._finalize(text, 1, used_ocr=True, engine=self.name)

    def _process_pdf(self, path: Path) -> OCRResult:
        embedded, pages = _extract_pdf_text(path)
        if len(embedded.strip()) >= _MIN_TEXT_CHARS:
            logger.info("PDF %s has embedded text (%d chars) — skipping OCR", path.name, len(embedded))
            return self._finalize(embedded, pages or 1, used_ocr=False, engine="pdf-text")

        # Scanned / image-only PDF
        if _ocrmypdf_available():
            try:
                text, pages2 = _ocr_pdf_via_ocrmypdf(path)
                return self._finalize(text, pages2 or pages or 1, used_ocr=True, engine="ocrmypdf+tesseract")
            except Exception as exc:
                logger.warning("OCRmyPDF failed for %s (%s) — trying image fallback", path.name, exc)

        if _pdf2image_available() and _tesseract_available():
            text, pages2 = _ocr_pdf_via_images(path)
            return self._finalize(text, pages2 or pages or 1, used_ocr=True, engine="pdf2image+tesseract")

        return OCRResult(
            text="",
            pages=pages or 0,
            ocr_engine=self.name,
            status="failed",
            error="Could not OCR this PDF. Install OCRmyPDF or poppler-utils.",
        )
