from __future__ import annotations

import logging
import os
import re
from pathlib import Path
from typing import Any

from app.core.postgres.erp_client import ErpDatabaseClient
from app.modules.inventory.application.sxd_ingestion.engine import OfficialSxdIngestionEngine
from app.modules.inventory.domain.ports.sxd_crawler_port import (
    ISxdCrawlerPort,
    SxdAnnouncement,
)
from app.modules.inventory.infrastructure.adapters.sxd_stealth_crawler_adapter import (
    SxdStealthCrawlerAdapter,
)

logger = logging.getLogger("dscons.inventory.application.crawl_and_sync_sxd_prices")


class CrawlAndSyncSxdPricesUseCase:
    """Application Use Case: Crawl latest official SXD material prices and synchronize into PostgreSQL.

    Strict Hexagonal Architecture & Ports/Adapters compliant:
    - Delegates all web crawling & anti-bot bypass to ISxdCrawlerPort.
    - Zero direct HTTP client (httpx/requests/urllib) imports in this file.
    - Orchestrates discovery, download to Price_Ref, and database ingestion.
    """

    def __init__(
        self,
        crawler: ISxdCrawlerPort | None = None,
        ingestion_engine: OfficialSxdIngestionEngine | None = None,
        price_ref_dir: str = "Price_Ref",
    ) -> None:
        self.crawler = crawler or SxdStealthCrawlerAdapter()
        self.price_ref_dir = Path(price_ref_dir)
        self.ingestion_engine = ingestion_engine or OfficialSxdIngestionEngine(
            price_ref_dir=str(self.price_ref_dir)
        )
        self.pg_client = ErpDatabaseClient()

    async def execute(self, max_announcements: int = 3) -> dict[str, Any]:
        """Execute end-to-end stealth crawl and database sync."""
        logger.info("Executing CrawlAndSyncSxdPricesUseCase (max_announcements=%d)...", max_announcements)

        # 1. Crawl announcements using stealth browser adapter
        announcements: list[SxdAnnouncement] = await self.crawler.crawl_latest_announcements(
            max_items=max_announcements
        )

        if not announcements:
            return {
                "status": "success",
                "message": "Không tìm thấy thông báo giá mới nào trên cổng Sở Xây dựng.",
                "crawled_count": 0,
                "downloaded_files": [],
                "upserted_count": 0,
            }

        downloaded_summary: list[dict[str, Any]] = []

        # 2. For each announcement, download files to Price_Ref/TB {number}
        for ann in announcements:
            # Clean folder name from announcement number, e.g. "622/TB-SXD" -> "TB 622"
            folder_name = "TB_NEW"
            if ann.announcement_number:
                m = re.search(r"(\d+)", ann.announcement_number)
                if m:
                    folder_name = f"TB {m.group(1)}"
            elif ann.period:
                folder_name = f"TB_{ann.period}"

            target_dir = self.price_ref_dir / folder_name
            target_dir.mkdir(parents=True, exist_ok=True)

            logger.info("Downloading attachments for '%s' to %s...", ann.title, target_dir)
            saved_paths = await self.crawler.download_announcement_files(ann, target_dir)

            downloaded_summary.append({
                "announcement_title": ann.title,
                "announcement_number": ann.announcement_number,
                "period": ann.period,
                "target_directory": str(target_dir),
                "files": {k: str(v) for k, v in saved_paths.items()},
            })

        # 3. Trigger ingestion engine to parse files in Price_Ref and upsert into PostgreSQL
        logger.info("Triggering official SXD ingestion engine into PostgreSQL...")
        upserted_count = self.ingestion_engine.ingest_to_postgres()

        return {
            "status": "success",
            "message": f"Đã thu thập thành công {len(announcements)} kỳ thông báo từ Sở Xây dựng và đồng bộ {upserted_count:,} bản ghi vào CSDL!",
            "crawled_count": len(announcements),
            "announcements": [ann.model_dump() for ann in announcements],
            "downloaded_files": downloaded_summary,
            "upserted_count": upserted_count,
        }
