"""OCR contracts — keep upload/review independent of the OCR engine."""

from __future__ import annotations

from dataclasses import dataclass, field
from typing import Any, Literal, Protocol

OCRStatus = Literal["pending", "completed", "failed", "unavailable"]


@dataclass
class ExtractedField:
    """Single invoice field with local OCR/extraction confidence (0.0–1.0)."""

    value: str | float | None
    confidence: float

    def to_dict(self) -> dict[str, Any]:
        return {"value": self.value, "confidence": round(self.confidence, 3)}


@dataclass
class ExtractionResult:
    """Deterministic invoice field extraction from OCR text."""

    supplier: ExtractedField
    invoice_number: ExtractedField
    invoice_date: ExtractedField
    due_date: ExtractedField
    subtotal: ExtractedField
    vat: ExtractedField
    total: ExtractedField
    currency: ExtractedField
    payment_ref: ExtractedField
    # Overall Local OCR/extraction confidence as 0–100 (matches existing UI).
    overall_confidence: float

    def to_dict(self) -> dict[str, Any]:
        return {
            "supplier": self.supplier.to_dict(),
            "invoice_number": self.invoice_number.to_dict(),
            "invoice_date": self.invoice_date.to_dict(),
            "due_date": self.due_date.to_dict(),
            "subtotal": self.subtotal.to_dict(),
            "vat": self.vat.to_dict(),
            "total": self.total.to_dict(),
            "currency": self.currency.to_dict(),
            "payment_ref": self.payment_ref.to_dict(),
            "overall_confidence": round(self.overall_confidence, 1),
        }


@dataclass
class OCRResult:
    text: str
    pages: int
    ocr_engine: str
    status: OCRStatus
    error: str | None = None
    used_ocr: bool = False
    extraction: ExtractionResult | None = None
    meta: dict[str, Any] = field(default_factory=dict)


class OCRProvider(Protocol):
    """Pluggable OCR + extraction backend."""

    name: str

    def available(self) -> bool:
        """True when the engine binaries/libraries are usable."""

    def process_document(self, file_path: str, mime_type: str | None) -> OCRResult:
        """Extract text and structured fields. Never modifies the original file."""
