from __future__ import annotations

import logging
import re
from pathlib import Path
from typing import Any

logger = logging.getLogger(__name__)


class DrawingPdfClassifierMixin:
    @staticmethod
    def classify_pdf_content_type(
        file_path: Path | str, max_check_pages: int = 5
    ) -> dict[str, Any]:
        """Tự động giám định sâu nội dung PDF (hình học trang, tỷ lệ vector nét vẽ, từ khóa văn bản)
        để phân biệt chính xác 100% giữa File Bản Vẽ Kỹ Thuật (Drawing) và File Hồ Sơ Mời Thầu / Dự Toán / Thuyết Minh (HSMT).
        """
        import unicodedata

        import pymupdf

        file_path = Path(file_path)
        if not file_path.exists() or file_path.suffix.lower() != ".pdf":
            return {
                "type": "unknown",
                "is_hsmt": False,
                "is_drawing": False,
                "score": 0.0,
                "reason": "Không phải file PDF",
            }

        # 1. Filename heuristic with normalized separators
        fn_lower = file_path.name.lower()

        # Normalize to NFD, strip diacritics, and convert 'đ' to 'd'
        fn_normalized = unicodedata.normalize("NFD", fn_lower)
        fn_stripped = "".join(
            c for c in fn_normalized if not unicodedata.combining(c)
        ).replace("đ", "d")

        clean_fn = re.sub(r"[\-_.]+", " ", fn_stripped)

        # Only non-accented keywords needed since we stripped diacritics
        hsmt_keywords_fn = [
            "tien luong",
            "du toan",
            "hsmt",
            "moi thau",
            "bieu gia",
            "bieu mau",
            "bieu",
            "bao cao ktkt",
            "thuyet minh",
            "bctm",
            "tong hop",
            "dutoan",
            "bang gia",
            "quyet dinh",
        ]
        drawing_keywords_fn = [
            "ban ve",
            "bvtc",
            "tktc",
            "cad",
            "drawing",
            "dwg",
            "blueprint",
            "ket cau",
            "chi tiet",
            "truc doc",
            "truc ngang",
            "binh do",
        ]

        fn_hsmt_match = any(kw in clean_fn for kw in hsmt_keywords_fn)
        fn_drawing_match = any(kw in clean_fn for kw in drawing_keywords_fn)

        try:
            doc = pymupdf.open(str(file_path))
            pages_to_check = min(len(doc), max_check_pages)
            if pages_to_check == 0:
                return {
                    "type": "drawing",
                    "is_hsmt": False,
                    "is_drawing": True,
                    "score": 0.5,
                    "reason": "PDF rỗng",
                }

            hsmt_points = 0.0
            drawing_points = 0.0
            reasons = []

            if fn_hsmt_match:
                hsmt_points += 5.0
                reasons.append(
                    "Tên file chứa từ khóa hồ sơ mời thầu / biểu mẫu / dự toán"
                )
            if fn_drawing_match:
                drawing_points += 5.0
                reasons.append("Tên file chứa từ khóa bản vẽ kỹ thuật")

            total_drawings_count = 0
            total_text_chars = 0
            landscape_pages_count = 0
            portrait_pages_count = 0

            hsmt_content_keywords = [
                "biểu mẫu mời thầu",
                "bảng kê hạng mục công việc",
                "khối lượng tham khảo",
                "chương iv",
                "mẫu số 01a",
                "mẫu số 01",
                "muasamcong.mpi.gov.vn",
                "e-gp",
                "cộng hòa xã hội chủ nghĩa việt nam",
                "hồ sơ mời thầu",
                "báo cáo kinh tế kỹ thuật",
                "thuyết minh",
                "bảng tổng hợp kinh phí",
                "tiên lượng",
                "đơn giá dự thầu",
                "đơn giá vật liệu",
                "khối lượng mời thầu",
                "bên mời thầu",
                "chủ đầu tư",
                "quyết định phê duyệt",
                "chương v",
            ]
            drawing_content_keywords = [
                "tỷ lệ 1:",
                "tl: 1/",
                "tỉ lệ 1:",
                "khung tên",
                "mặt cắt a-a",
                "mặt cắt 1-1",
                "mặt cắt b-b",
                "trắc dọc cống",
                "trắc ngang cống",
                "bình đồ tuyến",
                "người vẽ",
                "chủ trì thiết kế",
                "bản vẽ số:",
            ]

            for p_idx in range(pages_to_check):
                page = doc[p_idx]
                w, h = page.rect.width, page.rect.height
                if w > h * 1.15:
                    landscape_pages_count += 1
                elif h > w * 1.15:
                    portrait_pages_count += 1

                # Extract text
                page_text = (page.get_text() or "").lower()
                total_text_chars += len(page_text)

                for kw in hsmt_content_keywords:
                    if kw in page_text:
                        hsmt_points += 3.0
                        reasons.append(f"Trang {p_idx + 1} có từ khóa HSMT: '{kw}'")

                for kw in drawing_content_keywords:
                    if kw in page_text:
                        drawing_points += 3.0
                        reasons.append(f"Trang {p_idx + 1} có từ khóa Bản vẽ: '{kw}'")

            # Aspect ratio & layout analysis
            if landscape_pages_count > portrait_pages_count:
                drawing_points += 1.5
            elif portrait_pages_count > landscape_pages_count:
                hsmt_points += 2.0
                reasons.append(
                    "Khổ giấy dọc (Portrait/A4 - đặc trưng Hồ sơ mời thầu/Dự toán)"
                )

            # Vector density vs text density analysis
            avg_text_chars_per_page = total_text_chars / pages_to_check

            # If there is very little text but it's landscape, it's almost certainly a drawing
            if avg_text_chars_per_page < 300 and landscape_pages_count >= 1:
                drawing_points += 3.0
                reasons.append("Ít text và khổ giấy ngang (Đặc trưng bản vẽ CAD)")

            if avg_text_chars_per_page > 500:
                hsmt_points += 3.0
                reasons.append(
                    f"Mật độ văn bản dạng đoạn/bảng cao ({int(avg_text_chars_per_page)} ký tự/trang)"
                )

            is_hsmt = hsmt_points > drawing_points
            result_type = "hsmt" if is_hsmt else "drawing"
            confidence = min(
                0.99,
                max(
                    0.65,
                    abs(hsmt_points - drawing_points)
                    / (hsmt_points + drawing_points + 0.1),
                ),
            )

            return {
                "type": result_type,
                "is_hsmt": is_hsmt,
                "is_drawing": not is_hsmt,
                "confidence": confidence,
                "hsmt_score": hsmt_points,
                "drawing_score": drawing_points,
                "reasons": reasons[:4],
                "summary_reason": "; ".join(reasons[:2])
                if reasons
                else "Dựa trên định dạng trang và nét vẽ",
            }
        except Exception as e:
            logger.warning(
                "[TAKEOFF] Lỗi phân loại nội dung PDF '%s': %s", file_path.name, e
            )
            is_hsmt = fn_hsmt_match and not fn_drawing_match
            return {
                "type": "hsmt" if is_hsmt else "drawing",
                "is_hsmt": is_hsmt,
                "is_drawing": not is_hsmt,
                "confidence": 0.70,
                "reason": f"Fallback filename check: {e}",
            }
