from __future__ import annotations

import json
import re
import unicodedata
from pathlib import Path
from typing import Any

from .config import MANIFEST_PATH, ROOT_DIR, TOP_LEVEL_CATEGORY_RULES


def strip_accents(text: str) -> str:
    return "".join(
        ch
        for ch in unicodedata.normalize("NFKD", text)
        if not unicodedata.combining(ch)
    )


def normalize_key(text: str) -> str:
    text = strip_accents(text)
    text = text.replace("đ", "d").replace("Đ", "D")
    text = re.sub(r"\s+", " ", text).strip()
    return text


def normalize_relative_path(path: Path | str) -> str:
    raw = str(path).replace("\\", "/")
    normalized = normalize_key(raw).lower()

    normalized = normalized.replace("&", " ")
    normalized = re.sub(r"[^a-z0-9/.\- ]+", " ", normalized)
    normalized = re.sub(r"\s+", " ", normalized)
    return normalized.strip()


def slugify(text: str) -> str:
    normalized = normalize_key(text).lower()
    normalized = re.sub(r"[^a-z0-9]+", "-", normalized)
    normalized = re.sub(r"-{2,}", "-", normalized).strip("-")
    return normalized or "unknown"


def classify_top_level_entry(name: str) -> str:
    normalized = normalize_key(name)
    for raw, category in TOP_LEVEL_CATEGORY_RULES.items():
        if normalize_key(raw) == normalized:
            return category
    return "unknown_mixed"


def collect_top_level_summary(root_dir: Path) -> list[dict[str, object]]:
    rows: list[dict[str, object]] = []
    for entry in sorted(root_dir.iterdir(), key=lambda item: item.name):
        if entry.name.startswith("."):
            continue
        category = classify_top_level_entry(entry.name)
        rows.append(
            {
                "entry_name": entry.name,
                "entry_type": "directory" if entry.is_dir() else "file",
                "category": category,
            }
        )
    return rows


def chunk_text(
    text: str, chunk_size: int = 1200, overlap: int = 150, max_chunks: int = 10
) -> list[str]:
    cleaned = " ".join(text.split())
    if not cleaned:
        return []
    chunks: list[str] = []
    start = 0
    while start < len(cleaned) and len(chunks) < max_chunks:
        end = min(len(cleaned), start + chunk_size)
        chunks.append(cleaned[start:end])
        if end >= len(cleaned):
            break
        start = max(0, end - overlap)
    return chunks


def load_text(path: Path) -> str:
    suffix = path.suffix.lower()

    # Tránh xử lý file dung lượng quá lớn làm nghẽn CPU
    try:
        if path.stat().st_size > 5_000_000:
            return f"[Tài liệu dự án kích thước lớn: {path.name} - Định dạng {suffix}]"
    except Exception:
        pass

    if suffix in {".xlsx", ".xlsm"}:
        try:
            import openpyxl

            wb = openpyxl.load_workbook(path, read_only=True, data_only=True)
            sheet_texts = []
            for sheet in wb.worksheets[:5]:
                rows_text = []
                for row_idx, row in enumerate(sheet.iter_rows(values_only=True)):
                    if row_idx > 80:
                        break
                    row_vals = [
                        str(val).strip()
                        for val in row
                        if val is not None and str(val).strip()
                    ]
                    if row_vals:
                        rows_text.append(" | ".join(row_vals))
                if rows_text:
                    sheet_texts.append(
                        f"[Bảng tính: {sheet.title}]\n" + "\n".join(rows_text)
                    )
            wb.close()
            text = "\n\n".join(sheet_texts)
            if text.strip():
                return text[:12000]
        except Exception:
            pass

    if suffix == ".docx":
        import zipfile

        fragments: list[str] = []
        try:
            with zipfile.ZipFile(path) as archive:
                xml_bytes = archive.read("word/document.xml")
            xml_text = xml_bytes.decode("utf-8", errors="ignore")
            for raw in xml_text.replace("</w:p>", "\n").split("<"):
                if raw.startswith("w:t") and ">" in raw:
                    fragments.append(raw.split(">", 1)[1])
            text = " ".join(
                fragment.strip() for fragment in fragments if fragment.strip()
            )
            if text.strip():
                return text[:12000]
        except Exception:
            pass

    if suffix == ".pdf":
        try:
            import pypdf

            reader = pypdf.PdfReader(str(path))
            pages_text = []
            for idx, page in enumerate(reader.pages[:8]):
                extracted = page.extract_text()
                if extracted and extracted.strip():
                    pages_text.append(f"[Trang {idx + 1}]\n{extracted.strip()}")
            text = "\n\n".join(pages_text)
            if text.strip():
                return text[:12000]
        except Exception:
            pass

    if suffix in {".txt", ".md", ".json"}:
        try:
            return path.read_text(encoding="utf-8", errors="ignore")[:12000]
        except Exception:
            pass

    return f"[Tài liệu dự án: {path.name} - Định dạng {suffix}]"


def write_manifest(manifest: dict[str, Any]) -> None:
    MANIFEST_PATH.parent.mkdir(parents=True, exist_ok=True)
    MANIFEST_PATH.write_text(
        json.dumps(manifest, ensure_ascii=False, indent=2), encoding="utf-8"
    )


def iter_project_files(project_root: Path) -> list[Path]:
    supported = {".pdf", ".docx", ".doc", ".txt", ".md", ".xlsx", ".xls", ".xlsm"}
    files: list[Path] = []
    for path in sorted(project_root.rglob("*")):
        if not path.is_file():
            continue
        if path.name.startswith(".") or path.name.startswith("~$"):
            continue
        if path.suffix.lower() not in supported:
            continue
        files.append(path)
    return files


def relative_to_root(path: Path) -> str:
    return str(path.relative_to(ROOT_DIR))
