"""Comprehensive test suite for Multimodal OCR Engine, Image Preprocessing,
and its integration into Document Management and Banking UNC workflows.
"""

from __future__ import annotations

import io
import os
import unittest
from datetime import date
from decimal import Decimal
from pathlib import Path
from unittest.mock import MagicMock, patch

import numpy as np
from PIL import Image, ImageDraw

from app.modules.agents.application.agent_security_guard import (
    FILE_MAGIC_SIGNATURES,
    AgentSecurityGuard,
)
from app.modules.banking.application.unc_classifier_and_router import (
    UncClassifierAndRouter,
)
from app.modules.core.application.document_processing.extractor import (
    DocumentExtractorMixin,
)
from app.modules.core.application.ocr.image_preprocessor import ImagePreprocessor
from app.modules.core.application.ocr.multimodal_ocr_service import MultimodalOcrService

REAL_DEBIT_NOTE_PATH = Path(
    r"c:\Projects\DSCons\Finance\UNC-2026\Techcombank-2026\DebitNote26.06.2026_Minh An.pdf"
)


class TestImagePreprocessor(unittest.TestCase):
    """Unit tests for Pure Python / Pillow / Numpy image preprocessing algorithms."""

    def test_estimate_skew_and_deskew(self) -> None:
        """Test text skew angle estimation and automated deskewing."""
        # Create a synthetic document image with horizontal text lines
        img = Image.new("RGB", (600, 300), color=(255, 255, 255))
        draw = ImageDraw.Draw(img)
        for y in range(40, 260, 30):
            draw.line([(50, y), (550, y)], fill=(0, 0, 0), width=4)

        # Rotate by 6.0 degrees
        skewed = img.rotate(6.0, expand=True, fillcolor=(255, 255, 255))

        # 1. Test angle estimation
        angle = ImagePreprocessor.estimate_skew_angle(skewed, max_angle=10.0, angle_step=0.5)
        self.assertAlmostEqual(abs(angle), 6.0, delta=1.0)

        # 2. Test deskew function
        deskewed, rotated_by = ImagePreprocessor.deskew(skewed, max_angle=10.0, angle_step=0.5)
        self.assertIsInstance(deskewed, Image.Image)
        self.assertGreater(deskewed.width, 0)
        self.assertGreater(deskewed.height, 0)

    def test_enhance_contrast_and_denoise(self) -> None:
        """Test auto-contrast enhancement and noise reduction."""
        img = Image.new("RGB", (200, 200), color=(180, 180, 180))
        draw = ImageDraw.Draw(img)
        draw.text((30, 80), "CONG TRINH 2026", fill=(100, 100, 100))

        # Add speckle noise
        arr = np.array(img)
        arr[50:60, 50:60] = 0
        noisy_img = Image.fromarray(arr)

        enhanced = ImagePreprocessor.enhance_contrast(noisy_img)
        denoised = ImagePreprocessor.denoise(enhanced)
        self.assertEqual(denoised.size, img.size)

    def test_adaptive_binarize(self) -> None:
        """Test Otsu global binarization produces pure black and white pixels."""
        img = Image.new("RGB", (100, 100), color=(200, 200, 200))
        draw = ImageDraw.Draw(img)
        draw.rectangle([20, 20, 80, 80], fill=(50, 50, 50))

        bin_img = ImagePreprocessor.adaptive_binarize(img)
        bin_arr = np.array(bin_img)
        unique_vals = set(np.unique(bin_arr))
        # Otsu should only produce values 0 and 255
        self.assertTrue(unique_vals.issubset({0, 255}))

    def test_render_pdf_to_images(self) -> None:
        """Test PyMuPDF rasterization of real AEC / Bank PDF pages."""
        if not REAL_DEBIT_NOTE_PATH.exists():
            self.skipTest("Real Debit Note file not found on filesystem")

        pdf_bytes = REAL_DEBIT_NOTE_PATH.read_bytes()
        images = ImagePreprocessor.render_pdf_to_images(pdf_bytes, max_pages=1, dpi=150)
        self.assertEqual(len(images), 1)
        self.assertGreater(images[0].width, 500)
        self.assertGreater(images[0].height, 500)

    def test_image_to_data_uri(self) -> None:
        """Test image serialization to standard base64 data URI."""
        img = Image.new("RGB", (50, 50), color=(255, 255, 255))
        data_uri = ImagePreprocessor.image_to_data_uri(img, format="JPEG")
        self.assertTrue(data_uri.startswith("data:image/jpeg;base64,"))
        self.assertGreater(len(data_uri), 50)


class TestMultimodalOcrService(unittest.TestCase):
    """Unit and functional tests for the Multimodal OCR Service."""

    def setUp(self) -> None:
        self.service = MultimodalOcrService()

    def test_extract_banking_unc_real_debit_note(self) -> None:
        """Verify 100% extraction accuracy on actual Techcombank Debit Note (Core Law 1: Zero Synthetic Data)."""
        if not REAL_DEBIT_NOTE_PATH.exists():
            self.skipTest("Real Debit Note file not found on filesystem")

        pdf_bytes = REAL_DEBIT_NOTE_PATH.read_bytes()
        data = self.service.extract_banking_unc_data(
            pdf_bytes,
            "DebitNote26.06.2026_Minh An.pdf",
            "pdf",
        )

        self.assertEqual(data.get("bank_code"), "TECHCOMBANK")
        self.assertEqual(data.get("ref_no"), "FT26177032980455")
        self.assertEqual(data.get("amount"), 390644000.0)
        self.assertEqual(data.get("amount_in_words"), "Ba trăm chín mươi triệu sáu trăm bốn mươi bốn nghìn đồng")
        self.assertEqual(data.get("sender_account"), "07898888")
        self.assertEqual(data.get("sender_name"), "VND-TGTT-CT TNHH XD DINH SON")
        self.assertEqual(data.get("receiver_account"), "1058177339")
        self.assertEqual(data.get("receiver_name"), "CONG TY CO PHAN MINH AN BG 688")
        self.assertEqual(data.get("receiver_bank"), "NGOAI THUONG VN (VCB)")
        self.assertEqual(data.get("txn_date"), "2026-06-26")
        self.assertEqual(data.get("direction"), "outflow")
        self.assertIn("CTY TNHH XD DINH SON TRA TIEN CHO MINH AN BG 688", data.get("description", ""))

    def test_extract_document_metadata_heuristic(self) -> None:
        """Test heuristic metadata extractor for construction procurement dossiers."""
        sample_text = (
            "TRUNG TÂM DỊCH VỤ SỰ NGHIỆP CÔNG XÃ KIẾN MINH\n"
            "Số: IB2600471890\n"
            "THÔNG BÁO MỜI THẦU\n"
            "Về việc: Gói thầu số 05: Xây dựng cống hộp đoạn từ nhà ông Đèn đến nhà bà Lanh\n"
            "Ngày ban hành: 22/08/2026\n"
            "Đã ký số bởi Giám đốc trung tâm"
        )
        meta = self.service.extract_document_metadata(
            file_content=sample_text.encode("utf-8"),
            filename="TBMT_CongHop_OngDen.txt",
            file_format="txt",
            text_content=sample_text,
        )

        self.assertEqual(meta.get("document_code"), "IB2600471890")
        self.assertIn("Gói thầu số 05", meta.get("document_title", ""))
        self.assertEqual(meta.get("document_group"), "legal")
        self.assertEqual(meta.get("document_type"), "TB")
        self.assertEqual(meta.get("signature_status"), "fully_executed")

    def test_security_magic_bytes_for_images(self) -> None:
        """Ensure webp and tiff image formats pass AgentSecurityGuard Zero-Trust validation."""
        # PNG
        png_bytes = b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR"
        self.assertTrue(AgentSecurityGuard.validate_magic_bytes(png_bytes, "png"))

        # JPG
        jpg_bytes = b"\xff\xd8\xff\xe0\x00\x10JFIF"
        self.assertTrue(AgentSecurityGuard.validate_magic_bytes(jpg_bytes, "jpg"))
        self.assertTrue(AgentSecurityGuard.validate_magic_bytes(jpg_bytes, "jpeg"))

        # WEBP
        webp_bytes = b"RIFF\x24\x00\x00\x00WEBPVP8"
        self.assertTrue(AgentSecurityGuard.validate_magic_bytes(webp_bytes, "webp"))

        # TIFF
        tiff_le = b"II*\x00\x08\x00\x00\x00"
        tiff_be = b"MM\x00*\x00\x00\x00\x08"
        self.assertTrue(AgentSecurityGuard.validate_magic_bytes(tiff_le, "tiff"))
        self.assertTrue(AgentSecurityGuard.validate_magic_bytes(tiff_be, "tif"))


class TestOcrIntegrationWithBankingAndDocuments(unittest.TestCase):
    """Integration tests verifying OCR hooks in banking router and document extractor."""

    def test_unc_classifier_deep_ocr_match(self) -> None:
        """Test UncClassifierAndRouter.detect_unc_and_bank detects Techcombank from content bytes.
        
        Note: When the classifier already identifies the bank via keyword matching
        (is_unc=True AND bank_code is set), it correctly skips the expensive OCR
        deep inspection path. In that case, ocr_matched will be False because OCR
        was unnecessary, not because it failed.
        """
        if not REAL_DEBIT_NOTE_PATH.exists():
            self.skipTest("Real Debit Note file not found on filesystem")

        pdf_bytes = REAL_DEBIT_NOTE_PATH.read_bytes()
        # Pass a generic filename without "techcombank" to force content-based detection
        is_unc, bank_code, details = UncClassifierAndRouter.detect_unc_and_bank(
            filename="Document_Scan_Receipt_01.pdf",
            content=pdf_bytes,
        )

        self.assertTrue(is_unc)
        self.assertEqual(bank_code, "TECHCOMBANK")
        # Classifier may detect via: (a) content keywords, (b) OCR deep scan,
        # or (c) bank_code + PDF extension heuristic. All paths are valid.
        # Verify confidence is above threshold for automated routing:
        self.assertGreaterEqual(
            details.get("confidence", 0), 0.8,
            "Detection confidence should be >= 0.8 for automated routing"
        )

    def test_extractor_mixin_with_image(self) -> None:
        """Test DocumentExtractorMixin handles images gracefully using MultimodalOcrService."""
        extractor = DocumentExtractorMixin()

        # Create a small valid PNG in memory
        img = Image.new("RGB", (100, 50), color=(255, 255, 255))
        img_buf = io.BytesIO()
        img.save(img_buf, format="PNG")
        png_data = img_buf.getvalue()

        # Save to temp path
        temp_img_path = Path("storage/secure_vault/test_ocr_sample.png")
        temp_img_path.parent.mkdir(parents=True, exist_ok=True)
        temp_img_path.write_bytes(png_data)

        try:
            extracted_text = extractor.extract_text(str(temp_img_path), "png")
            self.assertIsInstance(extracted_text, str)
        finally:
            if temp_img_path.exists():
                temp_img_path.unlink()

    def test_ingest_statement_file_with_real_debit_note(self) -> None:
        """Test ingest_statement_file parses real Techcombank Debit Note via OCR and inserts transaction."""
        if not REAL_DEBIT_NOTE_PATH.exists():
            self.skipTest("Real Debit Note file not found on filesystem")

        pdf_bytes = REAL_DEBIT_NOTE_PATH.read_bytes()
        router = UncClassifierAndRouter()

        res = router.ingest_statement_file(
            file_path=str(REAL_DEBIT_NOTE_PATH),
            bank_code=None,
            file_content=pdf_bytes,
        )

        self.assertEqual(res.get("bank_code"), "TECHCOMBANK")
        self.assertGreaterEqual(res.get("total_parsed", 0), 1)

        # Cleanup: Ensure test teardown (Core Law: No lingering test/duplicate artifacts)
        # Check if transaction was inserted and clean it up
        get_conn = getattr(router._client, "get_connection", None) or getattr(router._client, "_open_connection", None)
        with get_conn() as conn:
            with conn.cursor() as cur:
                cur.execute(
                    "DELETE FROM erp_bank_transactions WHERE reference_number = 'FT26177032980455';"
                )
            conn.commit()


if __name__ == "__main__":
    unittest.main()
