"""Enhanced Multimodal OCR Engine.

Provides unified Multimodal Vision OCR and document understanding across DSCons ERP:
1. Document Management: Scanned AEC construction files, BBNT photos, legal dossiers.
2. Financial & Banking: Payment Orders (UNC), Debit Notes, Credit Notes, bank vouchers.
"""

from __future__ import annotations

import io
import json
import logging
import os
import re
from datetime import datetime
from decimal import Decimal
from typing import Any

import json_repair

from app.core.settings import Settings, get_settings
from app.modules.agents.application.agent_security_guard import AgentSecurityGuard
from app.modules.agents.application.llm_client import LLMClient
from app.modules.core.application.ocr.image_preprocessor import ImagePreprocessor

logger = logging.getLogger("dscons.ocr.multimodal_service")


class MultimodalOcrService:
    """Enterprise Multimodal OCR and document intelligence orchestrator."""

    def __init__(
        self,
        settings: Settings | None = None,
        llm_client: LLMClient | None = None,
    ) -> None:
        self._settings = settings or get_settings()
        self._llm_client = llm_client or LLMClient(self._settings)

    def extract_full_text(
        self,
        file_content: bytes,
        filename: str,
        file_format: str,
    ) -> str:
        """Extracts complete plain text from PDF or image formats using Hybrid OCR."""
        fmt = file_format.lower().strip()

        # 1. Plain text and DOCX fallback
        if fmt == "txt":
            try:
                return file_content.decode("utf-8")
            except Exception:
                return file_content.decode("latin1", errors="ignore")

        # 2. PDF Processing
        if fmt == "pdf":
            try:
                import pymupdf
                doc = pymupdf.open(stream=file_content, filetype="pdf")
                pages_text = []
                for pno in range(len(doc)):
                    p_text = doc[pno].get_text("text").strip()
                    if p_text:
                        pages_text.append(p_text)
                doc.close()

                combined = "\n\n".join(pages_text)
                # If digital PDF text is sufficient (> 50 chars), return immediately (Fast Path)
                if len(combined) > 50:
                    return combined
            except Exception as e:
                logger.warning("Digital PDF text extraction warning: %s", e)

            # Scanned PDF without text layer -> Render first 3 pages and transcribe via Vision
            images = ImagePreprocessor.render_pdf_to_images(file_content, max_pages=3, dpi=200)
            if images:
                return self._transcribe_images(images, filename)
            return ""

        # 3. Raster Image Processing (PNG, JPG, JPEG, WEBP, TIFF)
        if fmt in ("png", "jpg", "jpeg", "webp", "tiff"):
            try:
                img = ImagePreprocessor.preprocess_document_image(file_content, apply_deskew=True)
                return self._transcribe_images([img], filename)
            except Exception as e:
                logger.error("Failed to preprocess and transcribe image %s: %s", filename, e)
                return ""

        return ""

    def _transcribe_images(self, images: list[Any], filename: str) -> str:
        """Transcribes text from preprocessed images using Multimodal Vision or heuristic OCR."""
        if not images:
            return ""

        prompt = (
            "Bạn là chuyên gia số hóa hồ sơ công trình và chứng từ tài chính.\n"
            "Hãy đọc và phiên âm (transcribe) chính xác toàn bộ chữ viết, số liệu, con dấu, "
            "tiêu đề và nội dung xuất hiện trong bức ảnh này. Giữ nguyên cấu trúc dòng và bảng nếu có.\n"
            "Chỉ trả về nội dung văn bản thuần túy, không thêm lời dẫn giải."
        )

        pages_text: list[str] = []
        for idx, img in enumerate(images):
            data_uri = ImagePreprocessor.image_to_data_uri(img, format="JPEG", quality=85)
            try:
                transcribed = self._call_vision_model(data_uri, prompt)
                if transcribed and len(transcribed.strip()) > 10:
                    pages_text.append(transcribed.strip())
            except Exception as e:
                logger.warning(
                    "Multimodal Vision API call failed for page %d of %s: %s",
                    idx + 1,
                    filename,
                    e,
                )

        if pages_text:
            return "\n\n--- HẾT TRANG ---\n\n".join(pages_text)

        # Fallback transcription: Return placeholder or basic OCR indication
        return f"[Văn bản hình ảnh số hóa từ {filename}]"

    def extract_document_metadata(
        self,
        file_content: bytes,
        filename: str,
        file_format: str,
        text_content: str | None = None,
    ) -> dict[str, Any]:
        """Extracts structured legal/construction metadata (Doc code, Title, Issuer, Signer, Status)."""
        raw_text = text_content or self.extract_full_text(file_content, filename, file_format)
        fmt = file_format.lower().strip()

        # If we have an image or scanned PDF and API key is available, attempt Direct Vision extraction
        if fmt in ("png", "jpg", "jpeg", "webp", "tiff") or (fmt == "pdf" and len(raw_text) < 100):
            try:
                images = (
                    ImagePreprocessor.render_pdf_to_images(file_content, max_pages=1, dpi=200)
                    if fmt == "pdf"
                    else [ImagePreprocessor.preprocess_document_image(file_content)]
                )
                if images:
                    data_uri = ImagePreprocessor.image_to_data_uri(images[0])
                    vision_meta = self._extract_metadata_via_vision(data_uri, filename)
                    if vision_meta and vision_meta.get("document_code"):
                        return vision_meta
            except Exception as e:
                logger.warning("Vision metadata extraction failed: %s", e)

        # Fallback to Heuristic rule-based extraction from text
        return self._extract_metadata_heuristic(raw_text, filename)

    def extract_banking_unc_data(
        self,
        file_content: bytes,
        filename: str,
        file_format: str,
        text_content: str | None = None,
    ) -> dict[str, Any]:
        """Extracts bank payment order (UNC) or debit/credit advice transactions."""
        fmt = file_format.lower().strip()
        raw_text = text_content or ""

        # Step 1: Extract text if not provided
        if not raw_text:
            if fmt == "pdf":
                try:
                    import pymupdf
                    doc = pymupdf.open(stream=file_content, filetype="pdf")
                    pages_text = [doc[pno].get_text("text") for pno in range(min(3, len(doc)))]
                    raw_text = "\n".join(pages_text)
                    doc.close()
                except Exception:
                    pass
            elif fmt in ("png", "jpg", "jpeg", "webp", "tiff"):
                raw_text = self.extract_full_text(file_content, filename, fmt)

        # Step 2: Run Heuristic Structured Matcher on extracted text (High accuracy for Vietnamese banks)
        heuristic_res = self._parse_unc_heuristic(raw_text, filename)
        if heuristic_res.get("amount") and heuristic_res.get("amount") > 0:
            return heuristic_res

        # Step 3: If heuristic insufficient and image/scan available, call Vision AI
        if fmt in ("png", "jpg", "jpeg", "webp", "tiff", "pdf"):
            try:
                images = (
                    ImagePreprocessor.render_pdf_to_images(file_content, max_pages=1, dpi=200)
                    if fmt == "pdf"
                    else [ImagePreprocessor.preprocess_document_image(file_content)]
                )
                if images:
                    data_uri = ImagePreprocessor.image_to_data_uri(images[0])
                    vision_unc = self._extract_unc_via_vision(data_uri, filename)
                    if vision_unc and vision_unc.get("amount"):
                        return vision_unc
            except Exception as e:
                logger.warning("Vision UNC extraction error: %s", e)

        return heuristic_res

    def _parse_unc_heuristic(self, text: str, filename: str) -> dict[str, Any]:
        """Deep heuristic parser for Vietnamese Bank Debit Notes, Credit Notes, and UNCs."""
        result: dict[str, Any] = {
            "ref_no": None,
            "txn_date": None,
            "amount": None,
            "amount_in_words": None,
            "sender_account": None,
            "sender_name": None,
            "receiver_account": None,
            "receiver_name": None,
            "receiver_bank": None,
            "bank_code": None,
            "description": None,
            "direction": "outflow",
        }

        if not text:
            return result

        clean_text = text.replace("\r", "\n")
        lines = [line.strip() for line in clean_text.splitlines() if line.strip()]

        # 1. Detect Bank Code
        upper_text = clean_text.upper()
        if "TECHCOMBANK" in upper_text or "KỸ THƯƠNG" in upper_text or "07898888" in upper_text:
            result["bank_code"] = "TECHCOMBANK"
            result["sender_account"] = "07898888"
            result["sender_name"] = "CÔNG TY TNHH XÂY DỰNG ĐỊNH SƠN"
        elif "ACB" in upper_text or "Á CHÂU" in upper_text or "13456888888" in upper_text:
            result["bank_code"] = "ACB"
            result["sender_account"] = "13456888888"
            result["sender_name"] = "CÔNG TY TNHH XÂY DỰNG ĐỊNH SƠN"
        elif "VPBANK" in upper_text or "VIỆT NAM THỊNH VƯỢNG" in upper_text or "8667898888" in upper_text:
            result["bank_code"] = "VPBANK"
            result["sender_account"] = "8667898888"
            result["sender_name"] = "CÔNG TY TNHH XÂY DỰNG ĐỊNH SƠN"
        elif "VIETCOMBANK" in upper_text or "NGOẠI THƯƠNG" in upper_text or "VCB" in upper_text:
            result["bank_code"] = "VIETCOMBANK"
        elif "BIDV" in upper_text or "ĐẦU TƯ VÀ PHÁT TRIỂN" in upper_text:
            result["bank_code"] = "BIDV"
        elif "VIETINBANK" in upper_text or "CÔNG THƯƠNG" in upper_text:
            result["bank_code"] = "VIETINBANK"
        else:
            result["bank_code"] = "OTHER"

        # 2. Transaction / Reference Number
        # Techcombank FT number: e.g. FT26177032980455
        ft_match = re.search(r"\b(FT\d{10,20})\b", clean_text)
        if ft_match:
            result["ref_no"] = ft_match.group(1)
        else:
            ref_match = re.search(r"(?i)(?:Số giao dịch|Transaction No|Số GD|Mã GD|Số bút toán|Ref No)[\s.:]+([A-Z0-9_\-]+)", clean_text)
            if ref_match:
                result["ref_no"] = ref_match.group(1).strip()

        # 3. Transaction Date
        date_match = re.search(r"(?i)(?:Ngày giao dịch|Transaction Date|Ngày GD|Ngày|Date)[\s.:]+(\d{1,2}[/-]\d{1,2}[/-]\d{4})", clean_text)
        if date_match:
            raw_d = date_match.group(1).replace("-", "/")
            try:
                dt = datetime.strptime(raw_d, "%d/%m/%Y")
                result["txn_date"] = dt.strftime("%Y-%m-%d")
            except Exception:
                pass
        if not result["txn_date"]:
            any_date = re.search(r"\b(\d{2}/\d{2}/202[4-9])\b", clean_text)
            if any_date:
                try:
                    dt = datetime.strptime(any_date.group(1), "%d/%m/%Y")
                    result["txn_date"] = dt.strftime("%Y-%m-%d")
                except Exception:
                    pass

        # 4. Amount parsing
        amt_patterns = [
            r"(?i)(?:Tổng số tiền|Total amount|Số tiền giao dịch|Amount|Số tiền bằng số)[\s\S]{1,50}?[:\s]-?\s*([0-9]{1,3}(?:[.,]\d{3})+(?:[.,]\d{1,2})?|[0-9]{4,15})",
            r"(?i)(?:Số tiền|Số tiền chuyển|Giao dịch)[\s.:]+([0-9]{1,3}(?:[.,]\d{3})+(?:[.,]\d{1,2})?|[0-9]{4,15})\s*(?:VND|VNĐ|đ)",
            r"-\s*([0-9]{1,3}(?:,\d{3})+)",
        ]
        for pat in amt_patterns:
            amt_match = re.search(pat, clean_text)
            if amt_match:
                raw_amt_str = amt_match.group(1).replace(",", "").replace(".", "").strip()
                try:
                    amt_val = Decimal(raw_amt_str)
                    if amt_val > 1000:
                        result["amount"] = float(amt_val)
                        break
                except Exception:
                    pass

        # 5. Amount in words
        words_match = re.search(r"(?i)(?:Số tiền bằng chữ|Amount in words)[\s.:]+([^\n\r]+)", clean_text)
        if words_match:
            result["amount_in_words"] = words_match.group(1).strip()

        # 6. Parse structured dual-column line layout (Techcombank / Standard Debit advice)
        for i, line in enumerate(lines):
            line_upper = line.upper()
            if "ACCOUNT NAME" in line_upper:
                if i + 2 < len(lines):
                    potential_sender = lines[i + 1]
                    potential_receiver = lines[i + 2]
                    if not result["sender_name"] or "DINH SON" in potential_sender.upper():
                        result["sender_name"] = potential_sender
                    if not any(header in potential_receiver.upper() for header in ["SỐ TÀI KHOẢN", "ACCOUNT NUMBER"]):
                        result["receiver_name"] = potential_receiver
            elif "ACCOUNT NUMBER" in line_upper:
                if i + 2 < len(lines):
                    acc1 = re.sub(r"[^0-9]", "", lines[i + 1])
                    acc2 = re.sub(r"[^0-9]", "", lines[i + 2])
                    if acc1 and not result["sender_account"]:
                        result["sender_account"] = acc1
                    if acc2 and acc2 != result["sender_account"]:
                        result["receiver_account"] = acc2
            elif line_upper in ("BANK", "NGÂN HÀNG/ BANK", "NGÂN HÀNG"):
                if i + 2 < len(lines):
                    b2 = lines[i + 2]
                    if not any(kw in b2.upper() for kw in ["CHI TIẾT", "TRANSACTION DETAILS", "GIAO DỊCH"]):
                        result["receiver_bank"] = b2
            elif "NỘI DUNG THANH TOÁN" in line_upper or "DESCRIPTION:" in line_upper:
                if i + 1 < len(lines):
                    potential_desc = lines[i + 1]
                    if not any(kw in potential_desc.upper() for kw in ["GIAO DỊCH VIÊN", "TELLER", "KIỂM SOÁT VIÊN", "SUPERVISOR"]):
                        result["description"] = potential_desc

        # 7. Receiver Name and Account fallback via Regex if still None
        if not result["receiver_name"] or result["receiver_name"] in ("Tên tài khoản", "Account name"):
            rec_name_match = re.search(
                r"(?i)(?:Beneficiaries' account name|Người nhận tiền|Đơn vị thụ hưởng|Tên người nhận|Tên người thụ hưởng|Tên tài khoản thụ hưởng)[\s.:]+([^\n\r]+)",
                clean_text,
            )
            if rec_name_match:
                result["receiver_name"] = rec_name_match.group(1).strip()
            else:
                comp_match = re.search(r"\b(CONG TY [A-Z0-9\s.,-]+|CÔNG TY [A-ZÀ-Ỹ0-9\s.,-]+)\b", clean_text)
                if comp_match:
                    c_name = comp_match.group(1).strip()
                    if "DINH SON" not in c_name.upper() and "ĐỊNH SƠN" not in c_name.upper():
                        result["receiver_name"] = c_name

        if not result["receiver_account"]:
            rec_acc_match = re.search(r"(?i)(?:Số tài khoản nhận|Tài khoản thụ hưởng|Account number)[\s.:]+(\d{8,20})", clean_text)
            if rec_acc_match:
                acc_val = rec_acc_match.group(1)
                if acc_val != result["sender_account"]:
                    result["receiver_account"] = acc_val

        # 8. Description fallback
        if not result["description"]:
            desc_match = re.search(r"(?i)(?:Nội dung thanh toán|Description|Nội dung|Lý do nộp)[\s.:]+([^\n\r]+)", clean_text)
            if desc_match:
                desc_str = desc_match.group(1).strip()
                desc_str = re.split(r"(?i)(?:Giao dịch viên|Kiểm soát viên|Teller|Supervisor|Phiếu này)", desc_str)[0].strip()
                result["description"] = desc_str

        # 9. Direction check
        if "BÁO CÓ" in upper_text or "CREDIT" in upper_text:
            result["direction"] = "inflow"
        else:
            result["direction"] = "outflow"

        return result

    def _extract_metadata_heuristic(self, text: str, filename: str) -> dict[str, Any]:
        """Heuristic metadata extractor for construction and administrative documents."""
        # 1. Document Code
        code_match = re.search(r"(?i)(?:Số|Số hiệu|Mã TBMT|Mã hiệu)[\s.:]+([A-Z0-9_\-./]+)", text)
        doc_code = code_match.group(1).strip() if code_match else f"DOC-{os.path.splitext(filename)[0][:12]}"

        # 2. Document Title
        title = os.path.splitext(filename)[0].replace("_", " ").title()
        title_match = re.search(r"(?i)(?:V/v|Về việc|Trích yếu)[\s.:]+([^\n\r]+)", text)
        if title_match:
            title = title_match.group(1).strip()

        # 3. Document Group and Type
        upper_text = text.upper()
        if any(k in upper_text for k in ["THÔNG BÁO MỜI THẦU", "HỒ SƠ MỜI THẦU", "HSMT", "TBMT"]):
            doc_group, doc_type = "legal", "TB"
        elif any(k in upper_text for k in ["QUYẾT ĐỊNH", "QUYET DINH"]):
            doc_group, doc_type = "legal", "QD"
        elif any(k in upper_text for k in ["HỢP ĐỒNG", "HOP DONG"]):
            doc_group, doc_type = "legal", "HD"
        elif any(k in upper_text for k in ["BIÊN BẢN", "NGHIỆM THU"]):
            doc_group, doc_type = "technical", "BB"
        elif any(k in upper_text for k in ["ỦY NHIỆM CHI", "BÁO NỢ", "BÁO CÓ", "SAO KÊ"]):
            doc_group, doc_type = "financial", "UNC"
        else:
            doc_group, doc_type = "admin", "CV"

        # 4. Signature & Seal status
        sig_status = "draft"
        if any(k in upper_text for k in ["ĐÃ KÝ", "KÝ BỞI", "ĐÃ PHÊ DUYỆT", "CHỮ KÝ SỐ", "CHỨNG THƯ SỐ"]):
            sig_status = "fully_executed"
        elif "GIÁM ĐỐC" in upper_text or "CHỦ TỊCH" in upper_text:
            sig_status = "signed_unsigned_seal"

        return {
            "document_code": doc_code,
            "document_title": title,
            "document_group": doc_group,
            "document_type": doc_type,
            "signature_status": sig_status,
            "issuer_name": "Công ty TNHH Xây Dựng Định Sơn",
            "partner_name": None,
            "signer_name": None,
            "issue_date": datetime.now().strftime("%Y-%m-%d"),
            "summary_content": f"Văn bản {title} bóc tách tự động.",
        }

    def _call_vision_model(self, data_uri: str, prompt: str) -> str:
        """Invokes Multimodal Vision LLM through unified Enterprise AI Gateway."""
        from app.core.ai.domain.models import AiVisionRequest
        from app.core.ai.infrastructure.gateway import get_ai_gateway

        gateway = get_ai_gateway()
        req = AiVisionRequest(
            prompt=prompt,
            data_uri=data_uri,
            canonical_model_id="gemini-3.8-flash",
            temperature=0.1,
            max_tokens=2000,
        )
        resp = gateway.analyze_vision_sync(req)
        return resp.content

    def _extract_metadata_via_vision(self, data_uri: str, filename: str) -> dict[str, Any]:
        """Calls Vision model to extract structured document metadata as JSON."""
        prompt = (
            "Trích xuất thông tin văn bản công trình/pháp lý từ ảnh thành JSON đúng định dạng sau:\n"
            "{\n"
            '  "document_code": "Số hiệu văn bản",\n'
            '  "document_title": "Trích yếu văn bản đầy đủ",\n'
            '  "document_group": "legal | technical | financial | admin",\n'
            '  "document_type": "TB | QD | HD | BB | CV | UNC",\n'
            '  "signature_status": "fully_executed | signed_unsigned_seal | draft",\n'
            '  "issuer_name": "Tên cơ quan/đơn vị ban hành",\n'
            '  "partner_name": "Tên đối tác nếu có",\n'
            '  "signer_name": "Người ký",\n'
            '  "issue_date": "YYYY-MM-DD",\n'
            '  "summary_content": "Tóm tắt ngắn gọn 2 câu"\n'
            "}\n"
            "Chỉ trả về JSON thuần túy."
        )
        content = self._call_vision_model(data_uri, prompt)
        parsed = json_repair.repair_json(content, return_objects=True)
        return parsed if isinstance(parsed, dict) else {}

    def _extract_unc_via_vision(self, data_uri: str, filename: str) -> dict[str, Any]:
        """Calls Vision model to extract structured UNC / Bank statement transaction details as JSON."""
        prompt = (
            "Trích xuất chi tiết Ủy nhiệm chi / Phiếu báo nợ ngân hàng từ ảnh thành JSON đúng cấu trúc sau:\n"
            "{\n"
            '  "ref_no": "Mã giao dịch / Transaction No (ví dụ: FT26...)",\n'
            '  "txn_date": "YYYY-MM-DD",\n'
            '  "amount": 1000000,\n'
            '  "amount_in_words": "Số tiền bằng chữ",\n'
            '  "sender_account": "Số tài khoản chuyển",\n'
            '  "sender_name": "Tên đơn vị chuyển",\n'
            '  "receiver_account": "Số tài khoản nhận",\n'
            '  "receiver_name": "Tên đơn vị nhận",\n'
            '  "receiver_bank": "Tên ngân hàng thụ hưởng",\n'
            '  "bank_code": "TECHCOMBANK | ACB | VPBANK | VIETCOMBANK | BIDV | OTHER",\n'
            '  "description": "Nội dung chuyển tiền",\n'
            '  "direction": "outflow | inflow"\n'
            "}\n"
            "Lưu ý: Trường amount là số nguyên hoặc thập phân (không chứa dấu phẩy). Chỉ trả về JSON."
        )
        content = self._call_vision_model(data_uri, prompt)
        parsed = json_repair.repair_json(content, return_objects=True)
        return parsed if isinstance(parsed, dict) else {}
