from __future__ import annotations

from pathlib import Path
from typing import Any

from .constants import (
    DEFAULT_INGEST_BATCH,
    DOC_FAMILY_BY_DOCUMENT_TYPE,
    EVIDENCE_ROLE_BY_DOCUMENT_TYPE,
    GOLD_REFERENCE_ROLE_BY_DOCUMENT_TYPE,
    PACKAGE_NAME_MAP,
    REQUIRED_FOR_BY_DOCUMENT_TYPE,
    RETRIEVAL_PRIORITY_BY_DOCUMENT_TYPE,
    STAGE_NORMALIZATION_MAP,
    TASK_TYPE_BY_DOCUMENT_GROUP,
    TASK_TYPE_BY_DOCUMENT_TYPE,
    TRAINING_VALUE_BY_DOCUMENT_TYPE,
    USED_FOR_TASKS_BY_DOCUMENT_TYPE,
)
from .text_utils import infer_folder_scope, infer_section_title, parse_decision_fields


def infer_document_type(entry: dict[str, Any]) -> str:
    file_name = str(entry.get("file_name", "")).lower()
    notes = str(entry.get("notes", "")).lower()
    raw_document_type = str(entry.get("document_type", "unknown"))
    if (
        file_name.startswith("bia ")
        or file_name.startswith("bia_")
        or " bia " in f" {file_name} "
    ):
        return "bia_ho_so"
    if (
        file_name.startswith("0. bia")
        or file_name.startswith("bìa ")
        or file_name.startswith("bỉa ")
    ):
        return "bia_ho_so"
    if "cover-only" in notes or "file bìa" in notes or "cover" in notes:
        return "bia_ho_so"
    if "3a thanh toan lan 1" in file_name or "3a thanh toán lần 1" in file_name:
        return "phu_luc_thanh_toan"
    return raw_document_type


def infer_stage(entry: dict[str, Any], document_type: str) -> str:
    raw_stage = str(entry.get("stage", "unknown"))
    if document_type == "hop_dong":
        return "contracting"
    if document_type in {"de_nghi_thanh_toan", "phu_luc_thanh_toan"}:
        return "payment"
    if document_type == "quyet_toan":
        return "settlement"
    if document_type in {"nghiem_thu", "nhat_ky", "ho_so_chat_luong"}:
        return "execution"
    if document_type in {
        "ke_hoach_lua_chon_nha_thau",
        "phe_duyet_ket_qua_lcnt",
        "don_nhan_thau",
    }:
        return "procurement"
    if document_type in {"quyet_dinh", "phap_ly"}:
        return "legal_setup"
    return STAGE_NORMALIZATION_MAP.get(raw_stage, raw_stage)


def infer_version_role(entry: dict[str, Any], document_type: str) -> str:
    if not entry.get("is_ingest_candidate", False):
        return "duplicate"
    if document_type == "mau_bieu_02a":
        return "template_form"
    if document_type in {"bia_ho_so", "noise"}:
        return "duplicate"
    if str(entry.get("status", "")) == "needs_verification":
        return "unknown"
    return "effective"


def infer_requires_ocr(entry: dict[str, Any]) -> bool:
    return str(entry.get("extension", "")).lower() == ".pdf"


def infer_package_label(entry: dict[str, Any]) -> str:
    package_label = entry.get("package_label")
    if package_label:
        return str(package_label)
    if entry.get("document_group") in {
        "project_overview",
        "technical_design_estimate",
        "legal_approvals",
    }:
        return "package_project_wide"
    return "package_unknown"


def build_file_metadata(
    entry: dict[str, Any], schema: dict[str, Any]
) -> dict[str, Any]:
    defaults = dict(schema.get("project_defaults", {}))
    mapping_hints = schema.get("mapping_hints", {})
    package_map = schema.get("package_code_map", {})
    relative_path = str(entry["relative_path"])
    source_filename = str(entry["file_name"])
    source_format = str(entry["extension"]).lower().lstrip(".")
    document_group = str(entry["document_group"])
    document_type = infer_document_type(entry)
    package_label = infer_package_label(entry)
    decision_no, issue_date = parse_decision_fields(source_filename)
    business_group = mapping_hints.get("document_group_to_business_group", {}).get(
        document_group, "unknown"
    )
    knowledge_type = mapping_hints.get("document_type_to_knowledge_type", {}).get(
        document_type, "internal_process"
    )
    task_type = TASK_TYPE_BY_DOCUMENT_TYPE.get(
        document_type
    ) or TASK_TYPE_BY_DOCUMENT_GROUP.get(document_group, "audit_inspection")
    version_role = infer_version_role(entry, document_type)
    gold_reference_role = GOLD_REFERENCE_ROLE_BY_DOCUMENT_TYPE.get(
        document_type, "optional"
    )
    if not entry.get("is_ingest_candidate", False):
        gold_reference_role = "exclude"

    metadata = {
        **defaults,
        "schema_version": schema.get(
            "schema_version", defaults.get("schema_version", "1.0")
        ),
        "raw_path": relative_path,
        "source_file": relative_path,
        "source_filename": source_filename,
        "source_format": source_format,
        "document_group": document_group,
        "document_type": document_type,
        "business_group": business_group,
        "knowledge_type": knowledge_type,
        "task_type": task_type,
        "stage": infer_stage(entry, document_type),
        "status": str(entry.get("status", "available")),
        "version_role": version_role,
        "is_gold_reference": True,
        "gold_reference_role": gold_reference_role,
        "is_ingest_candidate": bool(entry.get("is_ingest_candidate", False)),
        "requires_ocr": infer_requires_ocr(entry),
        "package_label": package_label,
        "package_code": package_map.get(
            package_label, package_map.get("package_unknown", "")
        ),
        "package_name": PACKAGE_NAME_MAP.get(
            package_label, PACKAGE_NAME_MAP["package_unknown"]
        ),
        "title": Path(source_filename).stem,
        "section_title": infer_section_title(source_filename, document_type),
        "folder_scope": infer_folder_scope(relative_path),
        "required_for": REQUIRED_FOR_BY_DOCUMENT_TYPE.get(document_type, []),
        "evidence_role": EVIDENCE_ROLE_BY_DOCUMENT_TYPE.get(document_type, "reference"),
        "dependency_on": [],
        "related_docs": [],
        "decision_no": decision_no,
        "issue_date": issue_date,
        "effective_date": issue_date,
        "payment_stage": "lan_1"
        if "lần 1" in source_filename.lower() or "lan 1" in source_filename.lower()
        else "",
        "consulting_role": "",
        "retrieval_priority": RETRIEVAL_PRIORITY_BY_DOCUMENT_TYPE.get(
            document_type, 60
        ),
        "training_value": TRAINING_VALUE_BY_DOCUMENT_TYPE.get(document_type, "medium"),
        "confidence_note": "Best-effort mapping từ manifest, tên file, vị trí thư mục và metadata contract.",
        "notes": str(entry.get("notes", "")),
        "ingest_batch": DEFAULT_INGEST_BATCH,
        "ingest_ready": bool(entry.get("is_ingest_candidate", False)),
        "used_for_tasks": USED_FOR_TASKS_BY_DOCUMENT_TYPE.get(
            document_type, ["retrieval_grounding"]
        ),
        "doc_family": DOC_FAMILY_BY_DOCUMENT_TYPE.get(document_type, "reference"),
    }

    consulting_role_map = {
        "package_design": "thiet_ke",
        "package_tham_tra": "tham_tra",
        "package_02_tvgs": "giam_sat",
        "package_03_tvqlda": "qlda",
    }
    if package_label in consulting_role_map:
        metadata["consulting_role"] = consulting_role_map[package_label]
    return metadata
