from __future__ import annotations

import hashlib
import logging
import os
import re
import uuid
import xml.etree.ElementTree as ET
import zipfile
from pathlib import Path

logger = logging.getLogger("dscons.document_processing.extractor")


class DocumentExtractorMixin:
    """Document file extraction and text normalization utilities."""

    @staticmethod
    def sanitize_document_title(title: str | None, fallback_filename: str = "") -> str:
        """Làm sạch trích yếu / tiêu đề văn bản, loại bỏ các từ khóa cảnh báo bảo mật, nhãn hệ thống và ký tự thừa."""
        if not title:
            clean_fallback = (
                Path(fallback_filename).stem
                if fallback_filename
                else "Văn bản công trình"
            )
            return clean_fallback.replace("_", " ").strip()

        cleaned = str(title).strip()

        # 1. Loại bỏ các thẻ nhãn XML / HTML giả mạo hoặc sót lại
        cleaned = re.sub(r"<!--.*?-->", "", cleaned, flags=re.DOTALL)
        cleaned = re.sub(r"</?[a-zA-Z0-9_\-]+>", "", cleaned)
        cleaned = (
            cleaned.replace("[FILTERED_TAG]", "").replace("<!--", "").replace("-->", "")
        )

        # 2. Loại bỏ các tiền tố / hậu tố / cụm từ nhiễu về bảo mật và nhãn phân loại hệ thống
        noise_patterns = [
            r"(?i)\(\s*[^)]*(?:bảo mật|zero-trust|untrusted|cảnh báo)[^)]*\)",  # Toàn bộ ngoặc đơn chứa từ khóa bảo mật
            r"(?i)\[\s*[^\]]*(?:bảo mật|zero-trust|untrusted|cảnh báo)[^\]]*\]",  # Toàn bộ ngoặc vuông chứa từ khóa bảo mật
            r"(?i)\b(cảnh báo bảo mật|cảnh báo an toàn|bảo mật zero-trust|zero-trust|untrusted_payload|untrusted payload)\b",
            r"(?i)\b(bảo mật|mật)\s*:\s*",
            r"(?i)\s+bảo mật\s*$",  # Chữ 'bảo mật' ở cuối chuỗi
            r"(?i)^\s*bảo mật\s+",  # Chữ 'bảo mật' ở đầu chuỗi
        ]
        for pat in noise_patterns:
            cleaned = re.sub(pat, " ", cleaned)

        # 2.1. Loại bỏ các cặp dấu ngoặc rỗng sót lại
        cleaned = re.sub(r"\(\s*\)", "", cleaned)
        cleaned = re.sub(r"\[\s*\]", "", cleaned)

        # 3. Chuẩn hóa khoảng trắng và dấu gạch nối/dấu phẩy ở mép
        cleaned = re.sub(r"\s+", " ", cleaned).strip(" -:,_")

        # 4. Nếu sau khi làm sạch chuỗi bị rỗng hoặc quá ngắn (< 3 ký tự)
        if len(cleaned) < 3:
            if fallback_filename:
                clean_fallback = Path(fallback_filename).stem
                cleaned = clean_fallback.replace("_", " ").strip()
            else:
                cleaned = "Văn bản công trình"

        return cleaned

    def save_file(
        self, file_content: bytes, filename: str
    ) -> tuple[str, str, int, str, str]:
        """Lưu trữ file vào Secure Storage Vault cô lập ngoài Web Root và kiểm tra Magic Bytes.

        Returns:
            tuple: (file_url, file_path, file_size_bytes, file_format, content_hash)
        """
        from app.modules.agents.application.agent_security_guard import AgentSecurityGuard

        safe_filename = os.path.basename(filename)
        file_format = (
            safe_filename.split(".")[-1].lower() if "." in safe_filename else "bin"
        )

        # 1. Kiểm tra Magic Bytes chống tấn công giả mạo đuôi mở rộng
        if not AgentSecurityGuard.validate_magic_bytes(file_content, file_format):
            raise ValueError(
                f"File '{safe_filename}' không vượt qua kiểm tra chữ ký nhị phân (Magic Bytes) cho định dạng {file_format}."
            )

        # 2. Lưu trữ trong kho bảo mật cách ly ngoài Web Root
        vault_dir = Path("storage/secure_vault")
        vault_dir.mkdir(parents=True, exist_ok=True)

        # 3. Mã hóa tên file vật lý (Content-Hash + UUID)
        full_content_hash = hashlib.sha256(file_content).hexdigest()
        vault_filename = f"vault_{uuid.uuid4().hex}_{full_content_hash[:12]}.vault"
        local_path = (vault_dir / vault_filename).resolve()

        # Chống tấn công Path Traversal
        if not str(local_path).startswith(str(vault_dir.resolve())):
            raise ValueError(
                "Đường dẫn lưu trữ không hợp lệ hoặc chứa ký tự vượt quyền."
            )

        with open(local_path, "wb") as f:
            f.write(file_content)

        file_size = len(file_content)
        # Đường dẫn logic stream bảo vệ qua RBAC
        file_url = f"/v1/erp/documents/vault/{vault_filename}"
        file_path_str = str(local_path)

        logger.info(
            "Securely vaulted file %s to %s (format: %s, size: %s bytes, hash: %s)",
            filename,
            file_path_str,
            file_format,
            file_size,
            full_content_hash[:12],
        )
        return file_url, file_path_str, file_size, file_format, full_content_hash

    def extract_text(self, file_path_str: str, file_format: str) -> str:
        """Extract plain text from the given file path based on the format."""
        file_path = Path(file_path_str)
        if not file_path.exists():
            return ""

        fmt = file_format.lower().strip()
        if fmt == "txt":
            try:
                return file_path.read_text(encoding="utf-8")
            except Exception as e:
                logger.error("Failed to read text file %s: %s", file_path_str, e)
                return ""

        elif fmt == "docx":
            return self._extract_docx_text(file_path)

        elif fmt == "pdf":
            text = self._extract_pdf_text(file_path)
            # If digital text extraction is empty or minimal (< 50 chars), trigger Multimodal OCR fallback
            if len(text.strip()) < 50:
                try:
                    from app.modules.core.application.ocr.multimodal_ocr_service import MultimodalOcrService

                    ocr_svc = MultimodalOcrService()
                    ocr_text = ocr_svc.extract_full_text(
                        file_path.read_bytes(), file_path.name, "pdf"
                    )
                    if ocr_text and len(ocr_text.strip()) > len(text.strip()):
                        return ocr_text
                except Exception as e:
                    logger.warning("Scanned PDF Multimodal OCR fallback error: %s", e)
            return text

        elif fmt in ("png", "jpg", "jpeg", "webp", "tiff", "tif"):
            try:
                from app.modules.core.application.ocr.multimodal_ocr_service import MultimodalOcrService

                ocr_svc = MultimodalOcrService()
                return ocr_svc.extract_full_text(
                    file_path.read_bytes(), file_path.name, fmt
                )
            except Exception as e:
                logger.error("Image OCR extraction failed for %s: %s", file_path, e)
                return ""

        return ""

    def _extract_docx_text(self, file_path: Path) -> str:
        """Extract text from a DOCX file natively without python-docx dependency."""
        try:
            with zipfile.ZipFile(file_path) as docx:
                xml_content = docx.read("word/document.xml")
                tree = ET.fromstring(xml_content)
                ns = {
                    "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
                }
                text_nodes = tree.findall(".//w:t", ns)
                return "".join([node.text for node in text_nodes if node.text])
        except Exception as e:
            logger.error("Failed to extract docx text for %s: %s", file_path, e)
            return f"Lỗi trích xuất file docx: {e}"

    def _extract_pdf_text(self, file_path: Path) -> str:
        """Extract text from a PDF file using pypdf with binary fallback."""
        try:
            import pypdf

            reader = pypdf.PdfReader(file_path)
            text = ""
            for page in reader.pages:
                t = page.extract_text()
                if t:
                    text += t + "\n"
            return text
        except ImportError:
            logger.warning(
                "pypdf is not installed, running regex fallback for %s", file_path
            )
            try:
                with open(file_path, "rb") as f:
                    # Limit buffer size to 2MB to prevent ReDoS/OOM on fallback
                    content = f.read(2 * 1024 * 1024)
                import re

                # Safe non-backtracking regex for PDF string literals
                strings = re.findall(b"\\([^()]{1,1000}\\)", content)
                text_parts = []
                for s in strings:
                    clean_s = s[1:-1]
                    try:
                        decoded = clean_s.decode("utf-8", errors="ignore")
                        if len(decoded) > 3 and all(
                            c.isprintable() or c.isspace() for c in decoded
                        ):
                            text_parts.append(decoded)
                    except Exception:
                        pass
                if text_parts:
                    return "\n".join(text_parts)
                return "File PDF binary (cần cài đặt thư viện pypdf để trích xuất đầy đủ văn bản)."
            except Exception as e:
                return f"Không thể trích xuất PDF: {e}"
        except Exception as e:
            logger.error("Failed to extract PDF text for %s: %s", file_path, e)
            return f"Lỗi trích xuất PDF: {e}"
