from __future__ import annotations

import logging
from pathlib import Path
from typing import Any

logger = logging.getLogger(__name__)


class DrawingPdfStructureMdMixin:
    @staticmethod
    def extract_ai_friendly_pdf_structure(
        file_path: Path | str, max_pages: int = 160
    ) -> dict[str, Any]:
        """Tự động phân tích sâu cấu trúc không gian (Spatial Layout), nhận diện bảng biểu và tạo tài liệu Markdown AI-Friendly."""
        import pymupdf

        from app.modules.takeoff.application.vietnamese_cad_font_transcoder import (
            VietnameseCadFontTranscoder,
        )

        file_path = Path(file_path)
        if not file_path.exists() or file_path.suffix.lower() != ".pdf":
            return {
                "structured_markdown": "",
                "tables_count": 0,
                "specs_count": 0,
                "title_blocks": [],
            }

        try:
            doc = pymupdf.open(str(file_path))
            total_pages = min(len(doc), max_pages)

            markdown_sections: list[str] = [
                f"# HỒ SƠ BẢN VẼ KỸ THUẬT: {file_path.name.upper()}",
                f"**Tổng số trang:** {len(doc)} trang (Đã trích xuất chuyên sâu {total_pages} trang trọng yếu)\n",
            ]
            all_tables: list[dict[str, Any]] = []
            specs_found: list[str] = []
            title_blocks: list[str] = []

            for page_idx in range(total_pages):
                page = doc[page_idx]
                page_w = page.rect.width
                page_h = page.rect.height

                page_md = [
                    f"## 📄 TRANG {page_idx + 1} (Kích thước: {int(page_w)}x{int(page_h)} pt)"
                ]

                # 1. Table Detection & Extraction to Markdown Table
                try:
                    tables = page.find_tables()
                    if tables and len(tables.tables) > 0:
                        page_md.append("### 📊 BẢNG BIỂU & THỐNG KÊ CHI TIẾT:")
                        for t_idx, tab in enumerate(tables):
                            extracted_rows = tab.extract()
                            if not extracted_rows or len(extracted_rows) < 2:
                                continue

                            # Clean and format table
                            header = [
                                VietnameseCadFontTranscoder.auto_transcode_cad_text(
                                    str(c or "").strip().replace("\n", " ")
                                )
                                for c in extracted_rows[0]
                            ]
                            header = [
                                c if c else f"Cột {i + 1}" for i, c in enumerate(header)
                            ]

                            table_md = [
                                "| " + " | ".join(header) + " |",
                                "| " + " | ".join(["---"] * len(header)) + " |",
                            ]
                            for row in extracted_rows[1:]:
                                clean_row = [
                                    VietnameseCadFontTranscoder.auto_transcode_cad_text(
                                        str(c or "").strip().replace("\n", " ")
                                    )
                                    for c in row
                                ]
                                while len(clean_row) < len(header):
                                    clean_row.append("")
                                table_md.append(
                                    "| " + " | ".join(clean_row[: len(header)]) + " |"
                                )

                            page_md.append("\n".join(table_md) + "\n")
                            all_tables.append(
                                {
                                    "page": page_idx + 1,
                                    "table_index": t_idx + 1,
                                    "markdown": "\n".join(table_md),
                                }
                            )
                except Exception as tab_err:
                    logger.debug(
                        "[TAKEOFF] Table extraction notice on page %d: %s",
                        page_idx + 1,
                        tab_err,
                    )

                # 2. Spatial Text Block Extraction (Sorted Top-to-Bottom, Left-to-Right)
                blocks = page.get_text("blocks") or []
                sorted_blocks = sorted(blocks, key=lambda b: (b[1] // 25, b[0]))

                title_texts = []
                notes_texts = []
                general_texts = []

                for b in sorted_blocks:
                    if len(b) < 5 or (len(b) >= 7 and b[6] != 0):
                        continue
                    b_text = b[4].strip()
                    if not b_text or len(b_text) < 2:
                        continue
                    clean_b = VietnameseCadFontTranscoder.auto_transcode_cad_text(
                        b_text
                    )
                    clean_l = clean_b.lower()
                    x0, y0, x1, y1 = b[0], b[1], b[2], b[3]

                    # Title block heuristic (Bottom-Right area or explicit keywords)
                    if (x0 > page_w * 0.45 and y0 > page_h * 0.6) or any(
                        k in clean_l
                        for k in [
                            "công trình:",
                            "dự án:",
                            "hạng mục:",
                            "thiết kế:",
                            "chủ đầu tư:",
                            "tỷ lệ:",
                            "ký hiệu:",
                            "bản vẽ số:",
                        ]
                    ):
                        title_texts.append(clean_b)
                    elif any(
                        k in clean_l
                        for k in [
                            "ghi chú",
                            "quy chuẩn",
                            "tiêu chuẩn",
                            "mác bê tông",
                            "cốt thép",
                            "cừ larsen",
                            "k95",
                            "k98",
                            "cb240",
                            "cb300",
                            "cb400",
                            "độ dốc",
                            "bảo dưỡng",
                        ]
                    ):
                        notes_texts.append(clean_b)
                        specs_found.append(f"Trang {page_idx + 1}: {clean_b}")
                    else:
                        general_texts.append(clean_b)

                if title_texts:
                    title_block_str = "\n".join(title_texts[:80])
                    page_md.append(
                        f"### 📐 KHUNG TÊN BẢN VẼ (TITLE BLOCK):\n```yaml\n{title_block_str}\n```\n"
                    )
                    title_blocks.append(f"Trang {page_idx + 1}: {title_block_str}")

                if notes_texts:
                    page_md.append(
                        "### 📋 GHI CHÚ KỸ THUẬT & VẬT LIỆU (SPECS & NOTES):\n"
                        + "\n".join([f"- {t}" for t in notes_texts[:120]])
                        + "\n"
                    )

                if general_texts:
                    sample_texts = general_texts[:120]
                    page_md.append(
                        "### 📏 CÁC THÔNG SỐ & KÝ HIỆU HÌNH HỌC:\n"
                        + "\n".join([f"- {t}" for t in sample_texts])
                        + "\n"
                    )

                markdown_sections.append("\n".join(page_md))

            doc.close()
            full_structured_md = "\n\n".join(markdown_sections)
            return {
                "structured_markdown": full_structured_md,
                "tables_count": len(all_tables),
                "specs_count": len(specs_found),
                "title_blocks": title_blocks,
            }
        except Exception as e:
            logger.warning(
                "[TAKEOFF] Error extracting AI-friendly PDF structure: %s", e
            )
            return {
                "structured_markdown": "",
                "tables_count": 0,
                "specs_count": 0,
                "title_blocks": [],
            }
