"""Module nạp kho tri thức pháp chế và định mức xây dựng 2026 vào Qdrant Vector Store."""

from __future__ import annotations

import asyncio
import re
from pathlib import Path
from typing import Any

from app.api.dependencies import get_orchestrator

LAW_BASE_DIR = Path("KNOWLEDGE_BASE/LAW")
SOURCE_CODE = "VIETNAM_LAW_2026"
SOURCE_NAME = "KNOWLEDGE_BASE/LAW"


def extract_law_metadata(file_path: Path) -> dict[str, Any]:
    """Trích xuất siêu dữ liệu từ file văn bản pháp quy."""
    parent_folder = file_path.parent.name.upper()
    stem = file_path.stem
    text = file_path.read_text(encoding="utf-8", errors="ignore")

    doc_type = "luat"
    if "NGHI_DINH" in parent_folder or "Nghi_Dinh" in stem:
        doc_type = "nghi_dinh"
    elif "THONG_TU" in parent_folder or "Thong_Tu" in stem:
        doc_type = "thong_tu"

    # Trích xuất số hiệu văn bản
    law_number = stem
    if "Luat_" in stem:
        match = re.search(r"Luat_(\d+_\d+_[A-Za-z0-9]+)", stem)
        if match:
            law_number = match.group(1).replace("_", "/")
    elif "Nghi_Dinh_" in stem:
        match = re.search(r"Nghi_Dinh_(\d+_\d+_[A-Za-z0-9]+)", stem)
        if match:
            law_number = match.group(1).replace("_", "/")
    elif "Thong_Tu_" in stem:
        match = re.search(r"Thong_Tu_(\d+_\d+_[A-Za-z0-9]+)", stem)
        if match:
            law_number = match.group(1).replace("_", "/")

    # Trích xuất ngày hiệu lực
    effective_date = "2026-07-01" if "2026" in stem or "135" in stem else "2021-01-01"
    if "01/07/2026" in text:
        effective_date = "2026-07-01"
    elif "15/02/2026" in text:
        effective_date = "2026-02-15"

    title_match = re.search(r"^#\s+(.+)$", text, re.MULTILINE)
    title = title_match.group(1).strip() if title_match else stem.replace("_", " ")

    task_type = "contract_management"
    if doc_type == "luat":
        task_type = "contract_management"
    elif "Chat_Luong" in stem or "207" in stem or "06" in stem:
        task_type = "audit_inspection"
    elif (
        "Chi_Phi" in stem
        or "206" in stem
        or "10" in stem
        or "11" in stem
        or "12" in stem
        or "Ca_May" in stem
    ):
        task_type = "technical_delivery"
    elif "Hop_Dong" in stem or "210" in stem:
        task_type = "contract_management"

    return {
        "doc_type": doc_type,
        "law_number": law_number,
        "title": title,
        "effective_date": effective_date,
        "jurisdiction": "Vietnam-AEC-2026",
        "task_type": task_type,
        "stem": stem,
        "content": text,
    }


def chunk_law_document(
    doc_meta: dict[str, Any], max_chunk_size: int = 1200, overlap: int = 150
) -> list[dict[str, Any]]:
    """Phân rã văn bản pháp quy theo mục / điều / khoản hợp lý."""
    text = doc_meta["content"]
    sections = re.split(r"(?:\n\s*##\s+|\n\s*###\s+)", text)
    chunks_payload: list[dict[str, Any]] = []

    chunk_idx = 1
    for sec in sections:
        cleaned_sec = sec.strip()
        if not cleaned_sec:
            continue

        if len(cleaned_sec) <= max_chunk_size:
            chunk_texts = [cleaned_sec]
        else:
            paragraphs = cleaned_sec.split("\n\n")
            chunk_texts = []
            curr = ""
            for p in paragraphs:
                p_clean = p.strip()
                if not p_clean:
                    continue
                if len(curr) + len(p_clean) < max_chunk_size:
                    curr += ("\n\n" if curr else "") + p_clean
                else:
                    if curr:
                        chunk_texts.append(curr)
                    curr = p_clean
            if curr:
                chunk_texts.append(curr)

        for chunk_text in chunk_texts:
            if not chunk_text.strip():
                continue
            meta = {
                "source_code": SOURCE_CODE,
                "source_repository": SOURCE_NAME,
                "document_type": f"law_{doc_meta['doc_type']}",
                "law_number": doc_meta["law_number"],
                "doc_type": doc_meta["doc_type"],
                "title": doc_meta["title"],
                "effective_date": doc_meta["effective_date"],
                "jurisdiction": doc_meta["jurisdiction"],
                "knowledge_type": "legal_standard_2026",
                "task_type": doc_meta["task_type"],
                "chunk_index": chunk_idx,
                "id": f"LAW::{doc_meta['doc_type']}::{doc_meta['stem']}::chunk-{chunk_idx}",
            }
            chunks_payload.append(
                {
                    "text": f"[{doc_meta['title']} - {doc_meta['law_number']}]\n{chunk_text}",
                    "metadata": meta,
                }
            )
            chunk_idx += 1

    return chunks_payload


async def ingest_all_laws() -> dict[str, Any]:
    """Quét toàn bộ thư mục KNOWLEDGE_BASE/LAW và nạp vào Qdrant."""
    orchestrator = get_orchestrator()
    law_files = list(LAW_BASE_DIR.rglob("*.md"))

    total_files = len(law_files)
    total_chunks = 0
    ingested_details = []

    for file_path in sorted(law_files):
        doc_meta = extract_law_metadata(file_path)
        chunks = chunk_law_document(doc_meta)
        if not chunks:
            continue

        inserted = await orchestrator.ingest_knowledge(
            task_type=doc_meta["task_type"],
            chunks=chunks,
            default_metadata={
                "source_code": SOURCE_CODE,
                "source_repository": SOURCE_NAME,
            },
        )
        total_chunks += inserted
        ingested_details.append(
            {
                "file": file_path.name,
                "title": doc_meta["title"],
                "law_number": doc_meta["law_number"],
                "doc_type": doc_meta["doc_type"],
                "task_type": doc_meta["task_type"],
                "chunks": inserted,
            }
        )

    return {
        "status": "success",
        "total_law_files": total_files,
        "total_chunks_ingested": total_chunks,
        "details": ingested_details,
    }


if __name__ == "__main__":
    import sys

    sys.stdout.reconfigure(encoding="utf-8")
    result = asyncio.run(ingest_all_laws())
    print(
        f"=== DA NAP {result['total_chunks_ingested']} CHUNKS PHAP CHE TU {result['total_law_files']} VAN BAN VAO QDRANT ==="
    )
    for item in result["details"]:
        print(
            f"  - [{item['doc_type'].upper()}] {item['title']} ({item['law_number']}): {item['chunks']} chunks -> task {item['task_type']}"
        )
