from __future__ import annotations

import asyncio
import logging
import random
import re
import ssl
from pathlib import Path
from typing import Any
from urllib.parse import urljoin, unquote

import httpx
from playwright.async_api import async_playwright

from app.modules.inventory.domain.ports.sxd_crawler_port import (
    ISxdCrawlerPort,
    SxdAnnouncement,
    SxdAttachmentInfo,
)

logger = logging.getLogger("dscons.infrastructure.adapters.sxd_crawler")

# AIHawk-inspired Stealth User-Agent and Headers
DEFAULT_DESKTOP_UA = (
    "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
    "(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
)

AIHAWK_STEALTH_JS = """
// 1. Mask navigator.webdriver
Object.defineProperty(navigator, 'webdriver', {
    get: () => undefined,
});

// 2. Inject realistic window.chrome runtime
window.chrome = {
    runtime: {},
    app: {},
    loadTimes: function() {},
    csi: function() {}
};

// 3. Emulate languages and realistic plugins
Object.defineProperty(navigator, 'languages', {
    get: () => ['vi-VN', 'vi', 'en-US', 'en'],
});

Object.defineProperty(navigator, 'plugins', {
    get: () => [1, 2, 3, 4, 5],
});
"""


class SxdStealthCrawlerAdapter(ISxdCrawlerPort):
    """Secondary Infrastructure Adapter for crawling Department of Construction (Sở Xây Dựng) data.

    Implements AIHawk-inspired browser stealth and anti-bot bypass techniques:
    1. Removes Chromium automation flags (--disable-blink-features=AutomationControlled).
    2. Deep prototype masking of navigator.webdriver and window.chrome runtime.
    3. Handles self-signed/government CA SSL certificate chains (ignore_https_errors).
    4. Human-like behavioral simulation (jitter delay, natural scrolling).
    5. Resilient streaming file downloader for government CDNs.
    """

    def __init__(
        self,
        base_domain: str = "https://soxaydung.haiphong.gov.vn",
        timeout_ms: int = 25000,
    ) -> None:
        self.base_domain = base_domain.rstrip("/")
        self.timeout_ms = timeout_ms

    async def _human_jitter(self, min_ms: int = 400, max_ms: int = 1200) -> None:
        """Simulate natural human hesitation/interaction delay."""
        delay = random.uniform(min_ms, max_ms) / 1000.0
        await asyncio.sleep(delay)

    async def crawl_latest_announcements(
        self,
        base_url: str = "https://soxaydung.haiphong.gov.vn/thong-bao-chi-so-gia",
        max_items: int = 5,
    ) -> list[SxdAnnouncement]:
        """Crawl the latest material price and price index announcements using Stealth Browser."""
        announcements: list[SxdAnnouncement] = []

        async with async_playwright() as p:
            logger.info("Launching AIHawk Stealth Chromium instance for SXD portal...")
            browser = await p.chromium.launch(
                headless=True,
                args=[
                    "--disable-blink-features=AutomationControlled",
                    "--disable-infobars",
                    "--no-sandbox",
                    "--disable-dev-shm-usage",
                    "--disable-web-security",
                    "--ignore-certificate-errors",
                ],
            )

            context = await browser.new_context(
                user_agent=DEFAULT_DESKTOP_UA,
                viewport={"width": 1920, "height": 1080},
                locale="vi-VN",
                timezone_id="Asia/Ho_Chi_Minh",
                ignore_https_errors=True,
                extra_http_headers={
                    "Accept-Language": "vi-VN,vi;q=0.9,en-US;q=0.8,en;q=0.7",
                    "Sec-Ch-Ua": '"Chromium";v="124", "Google Chrome";v="124", "Not-A.Brand";v="99"',
                    "Sec-Ch-Ua-Mobile": "?0",
                    "Sec-Ch-Ua-Platform": '"Windows"',
                },
            )

            # Apply AIHawk deep stealth script to context
            await context.add_init_script(AIHAWK_STEALTH_JS)

            page = await context.new_page()

            try:
                logger.info("Navigating to %s...", base_url)
                response = await page.goto(
                    base_url, timeout=self.timeout_ms, wait_until="domcontentloaded"
                )
                await self._human_jitter()

                # Simulate human scroll down to trigger content rendering
                await page.evaluate("window.scrollBy(0, 400);")
                await self._human_jitter(300, 700)

                # Extract announcement article cards/links
                raw_articles = await page.evaluate(
                    """() => {
                    const links = Array.from(document.querySelectorAll("a"));
                    const items = [];
                    for (const a of links) {
                        const href = a.getAttribute("href") || "";
                        const title = (a.innerText || "").trim();
                        if (!href || !title) continue;
                        
                        const titleLower = title.toLowerCase();
                        if (
                            (titleLower.includes("giá") || titleLower.includes("vật liệu") || titleLower.includes("chỉ số")) &&
                            (titleLower.includes("thông báo") || titleLower.includes("công bố") || titleLower.includes("quyết định"))
                        ) {
                            items.push({
                                title: title,
                                href: href
                            });
                        }
                    }
                    return items;
                }"""
                )

                # Deduplicate by href
                seen_hrefs = set()
                candidate_items = []
                for item in raw_articles:
                    h = item["href"]
                    if h not in seen_hrefs:
                        seen_hrefs.add(h)
                        candidate_items.append(item)

                logger.info(
                    "Found %d candidate price announcement links on SXD portal",
                    len(candidate_items),
                )

                # Process top candidates up to max_items
                for candidate in candidate_items[:max_items]:
                    raw_title = candidate["title"]
                    rel_url = candidate["href"]
                    full_url = urljoin(self.base_domain, rel_url)

                    # Extract metadata from title
                    # e.g.: "Thông báo công bố giá vật liệu xây dựng tháng 8 năm 2026 - Thông báo số 622/TB-SXD ngày 09/9/2026"
                    doc_num_match = re.search(
                        r"(\d+[\w\-]*/(?:TB|QĐ|QD)-SXD)", raw_title, re.IGNORECASE
                    )
                    doc_num = doc_num_match.group(1) if doc_num_match else ""

                    date_match = re.search(
                        r"ngày\s*(\d{1,2}[\/\-\.]\d{1,2}[\/\-\.]\d{4})",
                        raw_title,
                        re.IGNORECASE,
                    )
                    publish_date = date_match.group(1) if date_match else ""

                    period_match = re.search(
                        r"tháng\s*(\d{1,2})(?:[\s\/\-]*(?:năm)?\s*(\d{4}))?",
                        raw_title,
                        re.IGNORECASE,
                    )
                    if period_match:
                        month = int(period_match.group(1))
                        year = int(period_match.group(2)) if period_match.group(2) else 2026
                        period = f"{year:04d}-{month:02d}"
                    else:
                        period = "2026-08"

                    # Now inspect the detail page to extract official attachments (PDF, XLSX)
                    attachments = await self._extract_detail_attachments(
                        context, full_url
                    )

                    announcements.append(
                        SxdAnnouncement(
                            title=raw_title,
                            announcement_number=doc_num,
                            publish_date=publish_date,
                            period=period,
                            source_url=full_url,
                            province="HAI_PHONG",
                            attachments=attachments,
                        )
                    )

            except Exception as exc:
                logger.error("Error during SXD stealth crawling: %s", exc, exc_info=True)
                raise
            finally:
                await browser.close()

        return announcements

    async def _extract_detail_attachments(
        self,
        context: Any,
        detail_url: str,
    ) -> list[SxdAttachmentInfo]:
        """Navigate to detail announcement page and extract all attachment download links."""
        page = await context.new_page()
        attachments: list[SxdAttachmentInfo] = []

        try:
            await page.goto(
                detail_url, timeout=self.timeout_ms, wait_until="domcontentloaded"
            )
            await self._human_jitter(200, 600)

            # Query all links pointing to documents/files
            raw_links = await page.evaluate(
                """() => {
                const links = Array.from(document.querySelectorAll("a[href]"));
                return links.map(a => ({
                    href: a.getAttribute("href") || "",
                    text: (a.innerText || "").trim(),
                    download: a.getAttribute("download") || ""
                }));
            }"""
            )

            for link in raw_links:
                href = link["href"]
                full_link = urljoin(self.base_domain, href)
                unquoted = unquote(full_link).lower()

                ext = None
                for candidate_ext in [".pdf", ".xlsx", ".xls", ".docx", ".doc"]:
                    if candidate_ext in unquoted:
                        ext = candidate_ext.lstrip(".")
                        break

                if ext:
                    # Determine filename
                    raw_filename = Path(href.split("?")[0]).name
                    clean_filename = unquote(raw_filename)
                    if not clean_filename or "." not in clean_filename:
                        clean_filename = f"attachment.{ext}"

                    attachments.append(
                        SxdAttachmentInfo(
                            filename=clean_filename,
                            download_url=full_link,
                            file_type=ext,
                            size_bytes=0,
                        )
                    )

        except Exception as exc:
            logger.warning(
                "Could not extract detail attachments from %s: %s", detail_url, exc
            )
        finally:
            await page.close()

        return attachments

    async def download_announcement_files(
        self,
        announcement: SxdAnnouncement,
        destination_dir: Path,
    ) -> dict[str, Path]:
        """Download announcement attachments with SSL bypass and anti-bot Referer."""
        destination_dir.mkdir(parents=True, exist_ok=True)
        saved_files: dict[str, Path] = {}

        # Set up SSL bypass and desktop headers
        headers = {
            "User-Agent": DEFAULT_DESKTOP_UA,
            "Referer": self.base_domain + "/",
            "Accept": "*/*",
        }

        async with httpx.AsyncClient(
            verify=False, headers=headers, timeout=60.0, follow_redirects=True
        ) as client:
            for att in announcement.attachments:
                dest_path = destination_dir / att.filename
                logger.info(
                    "Downloading SXD file %s from %s...", att.filename, att.download_url
                )

                try:
                    response = await client.get(att.download_url)
                    response.raise_for_status()

                    with open(dest_path, "wb") as f:
                        f.write(response.content)

                    # Identify role of file (pl1, pl2, tb)
                    fn_lower = att.filename.lower()
                    if "phu-luc-1" in fn_lower or "phu_luc_1" in fn_lower:
                        saved_files["pl1_pdf"] = dest_path
                    elif "phu-luc-2" in fn_lower or "phu_luc_2" in fn_lower or fn_lower.endswith(".xlsx"):
                        saved_files["pl2_excel"] = dest_path
                    elif "tb-" in fn_lower or "thong-bao" in fn_lower:
                        saved_files["main_doc_pdf"] = dest_path
                    else:
                        saved_files[att.filename] = dest_path

                    logger.info("Successfully downloaded and saved to %s", dest_path)
                except Exception as exc:
                    logger.error(
                        "Failed to download attachment %s: %s", att.download_url, exc
                    )

        return saved_files
