from __future__ import annotations

import json
from datetime import UTC, datetime
from pathlib import Path
from typing import Any

from app.api.dependencies import get_orchestrator

from .constants import (
    EXPECTED_DOCUMENT_TYPES,
    INGEST_BATCH,
    MANIFEST_PATH,
    PROJECT_CANDIDATES,
    SOURCE_CODE,
    SOURCE_NAME,
)
from .discovery import extract_project_root
from .helpers import chunk_text, load_text, slugify
from .models import ProjectCandidate, ProjectDocument


def build_chunk_payload(document: ProjectDocument) -> list[dict[str, Any]]:
    text = load_text(document.absolute_path)
    chunks = chunk_text(text)
    if not chunks:
        return []

    file_slug = slugify(document.absolute_path.stem)
    payload: list[dict[str, Any]] = []
    for index, chunk in enumerate(chunks, start=1):
        metadata = {
            "project_code": document.project_code,
            "project_name": document.project_name,
            "task_type": document.task_type,
            "doc_stage": document.doc_stage,
            "doc_family": document.doc_family,
            "knowledge_type": document.knowledge_type,
            "document_type": document.document_type,
            "required_for": list(document.required_for),
            "issuer": document.issuer,
            "signer": document.signer,
            "decision_no": document.decision_no,
            "issue_date": document.issue_date,
            "effective_date": document.effective_date,
            "source_code": SOURCE_CODE,
            "source_repository": SOURCE_NAME,
            "source_file": str(document.absolute_path),
            "relative_path": document.relative_path,
            "source_filename": document.source_filename,
            "source_format": document.source_format,
            "section_title": document.section_title,
            "folder_scope": document.folder_scope,
            "chunk_index": index,
            "chunk_count": len(chunks),
            "classification_source": document.classification_source,
            "matched_rule": document.matched_rule,
            "classification_confidence": document.confidence,
            "project_root": document.project_root,
            "ingest_batch": INGEST_BATCH,
            "tags": list(document.tags),
            "related_docs": list(document.related_docs),
            "compliance_refs": list(document.compliance_refs),
            "readiness_weight": document.readiness_weight,
            "required_for_payment": "payment" in document.required_for,
            "required_for_finalization": "finalization" in document.required_for,
            "required_for_commencement": "pre_award" in document.required_for
            or "contracting" in document.required_for,
            "draft_status": "unknown",
            "id": f"{document.project_code}::{document.document_type}::{file_slug}::chunk-{index}",
        }
        payload.append({"text": chunk, "metadata": metadata})
    return payload


def build_project_coverage_manifest(
    candidate: ProjectCandidate,
    documents: list[ProjectDocument],
    ingest_summary: dict[str, Any],
    project_root: Path | None,
) -> dict[str, Any]:
    document_type_counts: dict[str, int] = {}
    sample_unknown_files: list[str] = []
    classified_files = 0
    unknown_files = 0

    for document in documents:
        document_type_counts[document.document_type] = (
            document_type_counts.get(document.document_type, 0) + 1
        )
        if document.document_type == "unknown":
            unknown_files += 1
            if len(sample_unknown_files) < 10:
                sample_unknown_files.append(document.relative_path)
        else:
            classified_files += 1

    discovered_doc_types = {
        doc_type for doc_type in document_type_counts if doc_type != "unknown"
    }
    missing_document_types = sorted(
        doc_type
        for doc_type in EXPECTED_DOCUMENT_TYPES
        if doc_type not in discovered_doc_types
    )

    return {
        "project_code": candidate.project_code,
        "project_name": candidate.project_name,
        "project_root": str(project_root) if project_root else None,
        "project_root_exists": bool(project_root and project_root.exists()),
        "total_files": len(documents),
        "extractable_files": len(documents),
        "classified_files": classified_files,
        "unknown_files": unknown_files,
        "knowledge_files_ingested": ingest_summary.get("files_ingested", 0),
        "knowledge_chunks_ingested": ingest_summary.get("chunks", 0),
        "document_type_counts": document_type_counts,
        "missing_document_types": missing_document_types,
        "sample_unknown_files": sample_unknown_files,
        "sample_skipped_files": ingest_summary.get("skipped_files", []),
        "missing_files": ingest_summary.get("missing_files", []),
        "ingested_document_types": ingest_summary.get("document_types", []),
        "stages": ingest_summary.get("stages", []),
    }


def write_manifest(manifest: dict[str, Any]) -> None:
    # Ghi manifest để tiện rà soát.
    MANIFEST_PATH.parent.mkdir(parents=True, exist_ok=True)
    MANIFEST_PATH.write_text(
        json.dumps(manifest, ensure_ascii=False, indent=2), encoding="utf-8"
    )


async def ingest_qdrant_documents(
    project_documents_by_project: dict[str, list[ProjectDocument]],
) -> dict[str, Any]:
    orchestrator = get_orchestrator()
    total_chunks = 0
    total_documents_ingested = 0
    per_file: list[dict[str, Any]] = []
    per_project: dict[str, dict[str, Any]] = {}

    TASK_TYPE_MAP = {
        "project_approval": "project_planning",
        "procurement_tender": "procurement_tender",
        "contract_management": "contract_management",
        "technical_delivery": "technical_delivery",
        "cost_management": "technical_delivery",
        "site_handover": "technical_delivery",
        "site_execution": "technical_delivery",
        "quality_management": "audit_inspection",
        "payment_support": "payment_settlement",
        "closeout_documentation": "dossier_review",
        "legal_compliance": "legal_compliance",
        "audit_inspection": "audit_inspection",
        "dossier_review": "dossier_review",
        "project_planning": "project_planning",
        "payment_settlement": "payment_settlement",
    }

    for p_idx, candidate in enumerate(PROJECT_CANDIDATES, start=1):
        documents = project_documents_by_project.get(candidate.project_code, [])
        project_summary = per_project.setdefault(
            candidate.project_code,
            {
                "project_name": candidate.project_name,
                "files_seen": len(documents),
                "files_ingested": 0,
                "chunks": 0,
                "document_types": set(),
                "stages": set(),
                "missing_files": [],
                "skipped_files": [],
            },
        )

        project_chunks_by_task: dict[str, list[dict[str, Any]]] = {}

        for document in documents:
            if document.document_type == "unknown":
                project_summary["skipped_files"].append(document.relative_path)
                per_file.append(
                    {
                        "file": document.relative_path,
                        "project_code": document.project_code,
                        "status": "unclassified",
                        "chunks": 0,
                        "document_type": document.document_type,
                    }
                )
                continue
            if not document.absolute_path.exists():
                project_summary["missing_files"].append(document.relative_path)
                per_file.append(
                    {
                        "file": document.relative_path,
                        "project_code": document.project_code,
                        "status": "missing",
                        "chunks": 0,
                        "document_type": document.document_type,
                    }
                )
                continue

            payload = build_chunk_payload(document)
            if payload:
                mapped_task_type = TASK_TYPE_MAP.get(
                    document.task_type, "dossier_review"
                )
                project_chunks_by_task.setdefault(mapped_task_type, []).extend(payload)
                project_summary["files_ingested"] += 1
                project_summary["document_types"].add(document.document_type)
                project_summary["stages"].add(document.doc_stage)
                total_documents_ingested += 1

            per_file.append(
                {
                    "file": document.relative_path,
                    "project_code": document.project_code,
                    "status": "ingested" if payload else "empty",
                    "chunks": len(payload),
                    "document_type": document.document_type,
                    "doc_stage": document.doc_stage,
                    "task_type": document.task_type,
                    "classification_source": document.classification_source,
                }
            )

        # Batch ingest to Qdrant per task_type
        proj_chunk_count = 0
        for mapped_task, chunks_batch in project_chunks_by_task.items():
            if chunks_batch:
                inserted = await orchestrator.ingest_knowledge(
                    task_type=mapped_task,
                    chunks=chunks_batch,
                    default_metadata={
                        "source_code": SOURCE_CODE,
                        "source_repository": SOURCE_NAME,
                        "ingest_batch": INGEST_BATCH,
                    },
                )
                proj_chunk_count += inserted
                total_chunks += inserted

        project_summary["chunks"] = proj_chunk_count
        print(
            f"[{p_idx:02d}/{len(PROJECT_CANDIDATES):02d}] {candidate.project_code}: {len(documents)} tệp ({project_summary['files_ingested']} hợp lệ) -> {proj_chunk_count} chunks nạp Qdrant.",
            flush=True,
        )

    normalized_projects = {
        project_code: {
            "project_name": summary["project_name"],
            "files_seen": summary["files_seen"],
            "files_ingested": summary["files_ingested"],
            "chunks": summary["chunks"],
            "document_types": sorted(summary["document_types"]),
            "stages": sorted(summary["stages"]),
            "missing_files": summary["missing_files"],
            "skipped_files": summary["skipped_files"],
        }
        for project_code, summary in sorted(per_project.items())
    }

    coverage_projects = [
        build_project_coverage_manifest(
            candidate,
            project_documents_by_project.get(candidate.project_code, []),
            normalized_projects.get(candidate.project_code, {}),
            extract_project_root(candidate),
        )
        for candidate in PROJECT_CANDIDATES
    ]
    manifest = {
        "source_code": SOURCE_CODE,
        "source_name": SOURCE_NAME,
        "source_path": "HĐ-2026",
        "ingest_batch": INGEST_BATCH,
        "generated_at": datetime.now(UTC).isoformat(),
        "projects_processed": len(PROJECT_CANDIDATES),
        "totals": {
            "documents_discovered": sum(
                len(items) for items in project_documents_by_project.values()
            ),
            "documents_ingested": total_documents_ingested,
            "chunks_ingested": total_chunks,
            "projects_with_documents": sum(
                1 for items in project_documents_by_project.values() if items
            ),
        },
        "projects": coverage_projects,
    }
    write_manifest(manifest)

    return {
        "ingest_batch": INGEST_BATCH,
        "documents_configured": sum(
            len(items) for items in project_documents_by_project.values()
        ),
        "documents_ingested": total_documents_ingested,
        "total_chunks": total_chunks,
        "manifest_path": str(MANIFEST_PATH),
        "projects": normalized_projects,
        "files": per_file,
    }


async def run_full_hd2026_ingest() -> dict[str, Any]:
    from .constants import PROJECT_CANDIDATES, build_document_type_rules
    from .discovery import discover_project_documents

    rules = build_document_type_rules()
    docs_by_project: dict[str, list[ProjectDocument]] = {}
    for candidate in PROJECT_CANDIDATES:
        docs = discover_project_documents(candidate, rules)
        docs_by_project[candidate.project_code] = docs
    return await ingest_qdrant_documents(docs_by_project)


if __name__ == "__main__":
    import asyncio
    import sys

    sys.stdout.reconfigure(encoding="utf-8")
    result = asyncio.run(run_full_hd2026_ingest())
    print(
        f"=== DA NAP {result['total_chunks']} CHUNKS TU {result['documents_ingested']} TAI LIEU VAO QDRANT ==="
    )
    print(f"=== MANIFEST: {result['manifest_path']} ===")
