from __future__ import annotations

import asyncio
import json
import sys
from pathlib import Path

PROJECT_ROOT = Path(__file__).resolve().parents[1]
if str(PROJECT_ROOT) not in sys.path:
    sys.path.insert(0, str(PROJECT_ROOT))

from app.agents.personas import PERSONA_REGISTRY
from app.api.dependencies import get_orchestrator

SOURCE_PATH = Path("docs/company-people-knowledge.md")
INGEST_BATCH = "people-knowledge-v1"


def chunk_text(text: str, chunk_size: int = 1400, overlap: int = 180) -> list[str]:
    normalized = " ".join(text.split())
    if not normalized:
        return []
    chunks: list[str] = []
    start = 0
    while start < len(normalized):
        end = min(len(normalized), start + chunk_size)
        chunks.append(normalized[start:end])
        if end >= len(normalized):
            break
        start = max(0, end - overlap)
    return chunks


def load_source_text(path: Path) -> str:
    return path.read_text(encoding="utf-8")


def build_persona_chunks(text: str) -> list[dict[str, object]]:
    chunks: list[dict[str, object]] = []
    shared_defaults = {
        "knowledge_type": "internal_process",
        "document_type": "company_people_profile",
        "source_file": str(SOURCE_PATH),
        "source_filename": SOURCE_PATH.name,
        "source_format": "markdown",
        "ingest_batch": INGEST_BATCH,
        "tags": ["people", "persona", "xung-ho", "company-profile"],
    }

    overview_text = (
        "Tri thức nền về nhân sự AI Agent, khách hàng/đối tác và nguyên tắc xưng hô của DSCons.\n\n"
        f"Nguồn đầy đủ:\n{text}"
    )
    for index, chunk in enumerate(chunk_text(overview_text), start=1):
        chunks.append(
            {
                "text": chunk,
                "metadata": {
                    **shared_defaults,
                    "id": f"people::overview::{index}",
                    "task_type": "project_planning",
                    "agent_scope": "minh",
                    "section_title": "Tổng quan tri thức con người và cách xưng hô",
                    "chunk_index": index,
                },
            }
        )

    for agent_code, persona in PERSONA_REGISTRY.items():
        persona_text = "\n".join(
            [
                f"Mã agent: {persona['agent_code']}",
                f"Tên: {persona['display_name']}",
                f"Tuổi: {persona['age']}",
                f"Vai trò: {persona['role_title']}",
                f"Phong cách giao tiếp: {persona['communication_style']}",
                f"Tông mặc định: {persona['default_tone']}",
                "Hướng dẫn xưng hô:",
                *[f"- {item}" for item in persona["pronoun_guidance"]],
                f"Năng lực chính: {', '.join(persona['primary_capabilities'])}",
                f"Escalation: {', '.join(persona['escalation_targets'])}",
                f"Mô tả: {persona['description']}",
            ]
        )
        for index, chunk in enumerate(
            chunk_text(persona_text, chunk_size=1000, overlap=120), start=1
        ):
            chunks.append(
                {
                    "text": chunk,
                    "metadata": {
                        **shared_defaults,
                        "id": f"people::persona::{agent_code}::{index}",
                        "task_type": persona["primary_capabilities"][0],
                        "agent_code": agent_code,
                        "agent_scope": agent_code,
                        "person_code": agent_code,
                        "person_name": persona["display_name"],
                        "role_title": persona["role_title"],
                        "section_title": f"Hồ sơ AI employee {persona['display_name']}",
                        "chunk_index": index,
                    },
                }
            )

    return chunks


async def main() -> None:
    orchestrator = get_orchestrator()
    source_text = load_source_text(SOURCE_PATH)
    chunks = build_persona_chunks(source_text)
    ingested = await orchestrator.ingest_knowledge(
        task_type="project_planning",
        chunks=chunks,
        default_metadata={
            "knowledge_type": "internal_process",
            "document_type": "company_people_profile",
            "source_file": str(SOURCE_PATH),
            "ingest_batch": INGEST_BATCH,
        },
    )
    print(
        json.dumps(
            {
                "status": "ok",
                "source": str(SOURCE_PATH),
                "chunks_prepared": len(chunks),
                "chunks_ingested": ingested,
                "personas": list(PERSONA_REGISTRY.keys()),
            },
            ensure_ascii=False,
            indent=2,
        )
    )


if __name__ == "__main__":
    asyncio.run(main())
