from __future__ import annotations

import re
from pathlib import Path

from .constants import ROOT_DIR, TOP_LEVEL_CATEGORY_RULES
from .helpers import normalize_key, normalize_relative_path
from .models import DocumentTypeRule, ProjectCandidate, ProjectDocument


def classify_top_level_entry(name: str) -> str:
    normalized = normalize_key(name)
    for raw, category in TOP_LEVEL_CATEGORY_RULES.items():
        if normalize_key(raw) == normalized:
            return category
    return "unknown_mixed"


def collect_top_level_summary(root_dir: Path) -> list[dict[str, object]]:
    rows: list[dict[str, object]] = []
    for entry in sorted(root_dir.iterdir(), key=lambda item: item.name):
        if entry.name.startswith("."):
            continue
        rows.append(
            {
                "entry_name": entry.name,
                "entry_type": "directory" if entry.is_dir() else "file",
                "category": classify_top_level_entry(entry.name),
            }
        )
    return rows


def validate_project_candidates(candidates: list[ProjectCandidate]) -> None:
    seen_codes: set[str] = set()
    duplicate_codes: set[str] = set()
    for candidate in candidates:
        if candidate.project_code in seen_codes:
            duplicate_codes.add(candidate.project_code)
        seen_codes.add(candidate.project_code)
    if duplicate_codes:
        raise ValueError(
            f"Duplicate project_code values found: {', '.join(sorted(duplicate_codes))}"
        )


def extract_project_root(candidate: ProjectCandidate) -> Path | None:
    match = re.search(r"Nguồn:\s*(.+)$", candidate.notes)
    if not match:
        return None
    source_path = match.group(1).strip()
    path = Path(source_path)
    return path if path.exists() else None


def iter_project_files(project_root: Path) -> list[Path]:
    supported = {".pdf", ".docx", ".doc", ".txt", ".md", ".xlsx", ".xls", ".xlsm"}
    files: list[Path] = []
    for path in sorted(project_root.rglob("*")):
        if not path.is_file():
            continue
        if path.name.startswith(".") or path.name.startswith("~$"):
            continue
        if path.suffix.lower() not in supported:
            continue
        files.append(path)
    return files


def classify_project_file(
    path: Path, project_root: Path, rules: list[DocumentTypeRule]
) -> tuple[str, str | None, float, DocumentTypeRule | None]:
    relative_path = path.relative_to(project_root)
    normalized_path = normalize_relative_path(relative_path)
    normalized_name = normalize_relative_path(path.name)
    normalized_parent = normalize_relative_path(relative_path.parent)

    for rule in rules:
        target = normalized_path
        if rule.match_scope == "filename":
            target = normalized_name
        elif rule.match_scope == "folder":
            target = normalized_parent
        if rule.pattern in target:
            confidence = 0.95 if rule.match_scope != "any" else 0.88
            return rule.document_type, rule.pattern, confidence, rule

    if path.suffix.lower() == ".pdf" and "bd" in normalized_name:
        for rule in rules:
            if rule.document_type == "ban_ve_thi_cong":
                return rule.document_type, "fallback:bd", 0.72, rule
    return "unknown", None, 0.0, None


def build_project_document(
    candidate: ProjectCandidate,
    project_root: Path,
    path: Path,
    rules: list[DocumentTypeRule],
) -> ProjectDocument:
    document_type, matched_rule, confidence, rule = classify_project_file(
        path, project_root, rules
    )
    relative_path = path.relative_to(ROOT_DIR)
    default_section_title = path.stem

    if rule is None:
        return ProjectDocument(
            project_code=candidate.project_code,
            project_name=candidate.project_name,
            project_root=str(project_root),
            absolute_path=path,
            relative_path=str(relative_path),
            source_filename=path.name,
            source_format=path.suffix.lower().lstrip("."),
            folder_scope=str(path.parent.relative_to(ROOT_DIR)),
            document_type=document_type,
            classification_source="rule_based_fallback",
            matched_rule=matched_rule,
            confidence=confidence,
            task_type="dossier_inventory",
            knowledge_type="unclassified_document",
            doc_stage="unclassified",
            doc_family="unknown",
            required_for=(),
            section_title=default_section_title,
            issuer="",
            signer="",
            tags=("unknown", "needs_review"),
            compliance_refs=("Cần rà soát và phân loại thủ công bổ sung.",),
            readiness_weight=0.0,
        )

    return ProjectDocument(
        project_code=candidate.project_code,
        project_name=candidate.project_name,
        project_root=str(project_root),
        absolute_path=path,
        relative_path=str(relative_path),
        source_filename=path.name,
        source_format=path.suffix.lower().lstrip("."),
        folder_scope=str(path.parent.relative_to(ROOT_DIR)),
        document_type=rule.document_type,
        classification_source="rule_based",
        matched_rule=matched_rule,
        confidence=confidence,
        task_type=rule.task_type,
        knowledge_type=rule.knowledge_type,
        doc_stage=rule.doc_stage,
        doc_family=rule.doc_family,
        required_for=rule.required_for,
        section_title=rule.section_title,
        issuer=rule.issuer,
        signer=rule.signer,
        tags=(rule.document_type, rule.doc_stage, rule.doc_family),
        compliance_refs=(rule.description,) if rule.description else (),
        readiness_weight=0.8 if rule.doc_stage == "contracting" else 0.7,
    )


def discover_project_documents(
    candidate: ProjectCandidate, rules: list[DocumentTypeRule]
) -> list[ProjectDocument]:
    project_root = extract_project_root(candidate)
    if project_root is None:
        return []
    return [
        build_project_document(candidate, project_root, path, rules)
        for path in iter_project_files(project_root)
    ]
