"""Unified Knowledge Base Ingestion Script for Construction Laws and Project Knowledge."""

from __future__ import annotations

import asyncio
import sys
from pathlib import Path

PROJECT_ROOT = Path(__file__).resolve().parents[1]
if str(PROJECT_ROOT) not in sys.path:
    sys.path.insert(0, str(PROJECT_ROOT))

from app.api.dependencies import get_orchestrator

KNOWLEDGE_ROOT = PROJECT_ROOT / "KNOWLEDGE_BASE"
INGEST_BATCH = "legal-and-project-kb-v1"


def chunk_text(text: str, chunk_size: int = 1200, overlap: int = 150) -> list[str]:
    normalized = " ".join(text.split())
    if not normalized:
        return []
    chunks: list[str] = []
    start = 0
    while start < len(normalized):
        end = min(len(normalized), start + chunk_size)
        chunks.append(normalized[start:end])
        if end >= len(normalized):
            break
        start = max(0, end - overlap)
    return chunks


async def ingest_unified_knowledge():
    print(f" Scanning KNOWLEDGE_BASE at: {KNOWLEDGE_ROOT}")
    orchestrator = get_orchestrator()

    md_files = list(KNOWLEDGE_ROOT.glob("**/*.md"))
    print(f" Found {len(md_files)} markdown legal / project documents.")

    all_chunks: list[dict[str, object]] = []

    for doc_path in md_files:
        content = doc_path.read_text(encoding="utf-8")
        rel_path = doc_path.relative_to(KNOWLEDGE_ROOT)

        # Determine category
        parts = rel_path.parts
        category = parts[0] if len(parts) > 1 else "PROJECT_KNOWLEDGE"
        sub_category = parts[1] if len(parts) > 2 else "GENERAL"

        doc_slug = doc_path.stem.lower().replace(" ", "_")
        text_chunks = chunk_text(content)

        print(
            f"-> Processing [{category}/{sub_category}] {doc_path.name}: {len(text_chunks)} chunks"
        )

        for idx, text in enumerate(text_chunks, start=1):
            chunk_id = f"kb::{category.lower()}::{doc_slug}::{idx}"
            metadata = {
                "id": chunk_id,
                "document_title": doc_path.stem.replace("_", " "),
                "category": category,
                "sub_category": sub_category,
                "source_file": str(doc_path),
                "relative_path": str(rel_path),
                "chunk_index": idx,
                "total_chunks": len(text_chunks),
                "knowledge_type": "legal_standard"
                if category == "LAW"
                else "project_document",
                "document_type": "vietnamese_law_standard",
                "ingest_batch": INGEST_BATCH,
                "tags": [
                    category.lower(),
                    sub_category.lower(),
                    "vietnam-construction-standard",
                ],
            }
            all_chunks.append(
                {
                    "text": text,
                    "metadata": metadata,
                }
            )

    if not all_chunks:
        print("No chunks to ingest.")
        return

    print(
        f"\n Sending {len(all_chunks)} chunks to Qdrant vector database via Knowledge Orchestrator..."
    )
    ingested_count = await orchestrator.ingest_knowledge(
        task_type="legal_compliance",
        chunks=all_chunks,
        default_metadata={
            "knowledge_type": "legal_standard",
            "ingest_batch": INGEST_BATCH,
        },
    )

    print(
        f"\n Successfully ingested {ingested_count} chunks into Qdrant Vector Store (Collection: dscons_knowledge)!"
    )


if __name__ == "__main__":
    asyncio.run(ingest_unified_knowledge())
