#!/usr/bin/env python3
"""
FREEEXILE DOCUMENTATION SYSTEM LINTER & AI CATALOG COMPILER
Audits the documentation suite against Docs-as-Code & Diátaxis 2.0 standards:
- Validates YAML frontmatter presence, schema, and mandatory fields.
- Enforces line count limits: Soft Cap <= 400 lines, Hard Cap <= 600 lines.
- Validates internal clickable file:/// links integrity.
- Compiles machine-readable catalog: docs/documentation_index.json.

Usage:
    python tools/lint/check_documentation_system.py [--strict] [--json-out <path>]
"""

from __future__ import annotations
import os
import sys
import re
import json
from dataclasses import dataclass, field, asdict
from typing import List, Dict, Optional, Any

# Ensure UTF-8 on Windows
if hasattr(sys.stdout, "reconfigure"):
    sys.stdout.reconfigure(encoding="utf-8")
if hasattr(sys.stderr, "reconfigure"):
    sys.stderr.reconfigure(encoding="utf-8")

PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "../.."))
DOCS_ROOT = os.path.join(PROJECT_ROOT, "docs")
INDEX_JSON_PATH = os.path.join(DOCS_ROOT, "documentation_index.json")

# Limits
DOC_SOFT_CAP = 400
DOC_HARD_CAP = 600

# Canonical Taxonomy
ALLOWED_CATEGORIES = {
    "architecture",
    "game_design",
    "client",
    "security",
    "operations",
    "standards",
    "adr",
}

ALLOWED_DIATAXIS = {
    "tutorial",
    "how-to",
    "reference",
    "explanation",
}

MANDATORY_FIELDS = [
    "doc_id",
    "title",
    "category",
    "diataxis_type",
    "owner_role",
    "tags",
    "summary",
]

EXCLUDE_DIRS = {
    "design_requests",
    "audit",
    ".git",
    "__pycache__",
}


@dataclass
class DocViolation:
    file_path: str
    severity: str  # "ERROR" | "WARNING"
    rule: str
    message: str


@dataclass
class DocEntry:
    doc_id: str
    title: str
    rel_path: str
    abs_path: str
    category: str
    diataxis_type: str
    status: str
    version: str
    owner_role: str
    last_updated: str
    tags: List[str]
    related_code: List[str]
    related_docs: List[str]
    summary: str
    line_count: int


def parse_simple_yaml_frontmatter(content: str) -> Optional[Dict[str, Any]]:
    """Lightweight deterministic YAML frontmatter parser (no heavy dependencies)."""
    if not content.startswith("---"):
        return None
    parts = content.split("---", 2)
    if len(parts) < 3:
        return None
    fm_raw = parts[1].strip()

    data: Dict[str, Any] = {}
    current_list_key: Optional[str] = None

    for line in fm_raw.splitlines():
        line_str = line.strip()
        if not line_str or line_str.startswith("#"):
            continue

        # Sub-list items: - "item"
        if line.startswith("  - ") or line.startswith("    - "):
            if current_list_key:
                val = line_str.lstrip("- ").strip().strip('"').strip("'")
                data.setdefault(current_list_key, []).append(val)
            continue

        if ":" in line_str:
            key, val = line_str.split(":", 1)
            key = key.strip()
            val = val.strip()

            # Check if list starts: tags: ["a", "b"]
            if val.startswith("[") and val.endswith("]"):
                inner = val[1:-1].strip()
                items = [x.strip().strip('"').strip("'") for x in inner.split(",") if x.strip()]
                data[key] = items
                current_list_key = None
            elif not val:  # Next lines are list items
                data[key] = []
                current_list_key = key
            else:
                data[key] = val.strip('"').strip("'")
                current_list_key = None

    return data


def audit_document(file_path: str, rel_path: str) -> tuple[Optional[DocEntry], List[DocViolation]]:
    violations: List[DocViolation] = []

    try:
        with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
            content = f.read()
            lines = content.splitlines()
            line_count = len(lines)
    except Exception as e:
        violations.append(DocViolation(rel_path, "ERROR", "IO_ERROR", f"Không thể đọc file: {e}"))
        return None, violations

    # Check line length
    if line_count > DOC_HARD_CAP:
        violations.append(
            DocViolation(
                rel_path,
                "ERROR",
                "HARD_CAP_EXCEEDED",
                f"Tài liệu vượt quá Hard Cap ({line_count} > {DOC_HARD_CAP} dòng). Bắt buộc phân tách module!",
            )
        )
    elif line_count > DOC_SOFT_CAP:
        violations.append(
            DocViolation(
                rel_path,
                "WARNING",
                "SOFT_CAP_EXCEEDED",
                f"Tài liệu dài ({line_count} > {DOC_SOFT_CAP} dòng). Khuyến nghị cô đọng bảng biểu.",
            )
        )

    # Parse YAML frontmatter
    fm = parse_simple_yaml_frontmatter(content)
    if fm is None:
        violations.append(
            DocViolation(
                rel_path,
                "ERROR",
                "MISSING_FRONTMATTER",
                "Tài liệu thiếu khối metadata YAML Frontmatter chuẩn ở đầu file (---).",
            )
        )
        return None, violations

    # Validate mandatory fields
    for field_name in MANDATORY_FIELDS:
        if field_name not in fm or not fm[field_name]:
            violations.append(
                DocViolation(
                    rel_path,
                    "ERROR",
                    "MISSING_FIELD",
                    f"Thiếu trường bắt buộc trong YAML frontmatter: '{field_name}'",
                )
            )

    cat = str(fm.get("category", "")).lower()
    if cat and cat not in ALLOWED_CATEGORIES:
        violations.append(
            DocViolation(
                rel_path,
                "WARNING",
                "INVALID_CATEGORY",
                f"Phân loại category '{cat}' không thuộc taxonomy chuẩn: {ALLOWED_CATEGORIES}",
            )
        )

    dia = str(fm.get("diataxis_type", "")).lower()
    if dia and dia not in ALLOWED_DIATAXIS:
        violations.append(
            DocViolation(
                rel_path,
                "WARNING",
                "INVALID_DIATAXIS",
                f"Phân loại Diátaxis '{dia}' không hợp lệ: {ALLOWED_DIATAXIS}",
            )
        )

    # Check clickable links integrity: [text](file:///c:/Projects/FreeExile/...)
    link_pattern = re.compile(r'\[([^\]]+)\]\(file:///([Cc]:/[^\)]+)\)')
    for match in link_pattern.finditer(content):
        raw_target = match.group(2)
        # Strip URL fragment (#L10-L20)
        file_part = raw_target.split("#")[0]
        # Normalize to Windows path
        norm_path = file_part.replace("/", os.sep)
        if not os.path.exists(norm_path):
            violations.append(
                DocViolation(
                    rel_path,
                    "WARNING",
                    "BROKEN_FILE_LINK",
                    f"Liên kết cục bộ không tồn tại: {match.group(0)}",
                )
            )

    entry = DocEntry(
        doc_id=str(fm.get("doc_id", "DOC-UNKNOWN")),
        title=str(fm.get("title", os.path.basename(file_path))),
        rel_path=rel_path.replace("\\", "/"),
        abs_path=f"file:///{file_path.replace(os.sep, '/')}",
        category=cat,
        diataxis_type=dia,
        status=str(fm.get("status", "canonical")),
        version=str(fm.get("version", "2026.1")),
        owner_role=str(fm.get("owner_role", "unassigned")),
        last_updated=str(fm.get("last_updated", "")),
        tags=fm.get("tags", []) if isinstance(fm.get("tags"), list) else [str(fm.get("tags"))],
        related_code=fm.get("related_code", []) if isinstance(fm.get("related_code"), list) else [],
        related_docs=fm.get("related_docs", []) if isinstance(fm.get("related_docs"), list) else [],
        summary=str(fm.get("summary", "")),
        line_count=line_count,
    )

    return entry, violations


def run_documentation_audit(strict: bool = False, json_out: Optional[str] = None) -> bool:
    print("=" * 80)
    print("       FREEEXILE DOCUMENTATION SYSTEM & AI CATALOG AUDIT (2026.1)")
    print("=" * 80)

    all_entries: List[DocEntry] = []
    all_violations: List[DocViolation] = []
    scanned_count = 0

    for root, dirs, files in os.walk(DOCS_ROOT):
        # Exclude directories
        dirs[:] = [d for d in dirs if d not in EXCLUDE_DIRS]

        for file in files:
            if not file.endswith(".md"):
                continue
            file_path = os.path.join(root, file)
            rel_path = os.path.relpath(file_path, PROJECT_ROOT)
            scanned_count += 1

            entry, viols = audit_document(file_path, rel_path)
            all_violations.extend(viols)
            if entry:
                all_entries.append(entry)

    errors = [v for v in all_violations if v.severity == "ERROR"]
    warnings = [v for v in all_violations if v.severity == "WARNING"]

    print(f"Tổng số tài liệu quét : {scanned_count}")
    print(f"Số tài liệu hợp lệ    : {len(all_entries)}")
    print(f"Lỗi nghiêm trọng (ERR): {len(errors)}")
    print(f"Cảnh báo (WARNING)   : {len(warnings)}")
    print("-" * 80)

    if errors:
        print("\n[!] PHÁT HIỆN LỖI NGHIÊM TRỌNG (ERROR):")
        for err in errors:
            print(f"  ❌ [{err.rule}] {err.file_path}: {err.message}")

    if warnings:
        print("\n[*] CÁC CẢNH BÁO (WARNING):")
        for w in warnings[:15]:  # Limit output
            print(f"  ⚠️ [{w.rule}] {w.file_path}: {w.message}")
        if len(warnings) > 15:
            print(f"  ... và còn {len(warnings) - 15} cảnh báo khác.")

    # Compile documentation_index.json
    catalog_data = {
        "version": "2026.1",
        "generated_at": "2026-09-29T20:00:00Z",
        "total_documents": len(all_entries),
        "documents": [asdict(e) for e in sorted(all_entries, key=lambda x: (x.category, x.doc_id))],
    }

    target_json = json_out if json_out else INDEX_JSON_PATH
    with open(target_json, "w", encoding="utf-8") as f:
        json.dump(catalog_data, f, ensure_ascii=False, indent=2)

    print(f"\n[✓] Đã biên dịch Machine-Readable AI Catalog: {target_json}")

    print("=" * 80)
    if errors or (strict and warnings):
        print("❌ KẾT QUẢ: AUDIT THẤT BẠI - HÃY KHẮC PHỤC CÁC LỖI TRÊN!")
        print("=" * 80)
        return False
    else:
        print("✅ KẾT QUẢ: HỆ THỐNG TÀI LIỆU ĐẠT CHUẨN DOCS-AS-CODE & AI-FRIENDLY 100%!")
        print("=" * 80)
        return True


if __name__ == "__main__":
    strict_mode = "--strict" in sys.argv
    success = run_documentation_audit(strict=strict_mode)
    sys.exit(0 if success else 1)
