#!/usr/bin/env python3
"""
FREEEXILE I18N HYGIENE & ANTI-REGRESSION STATIC LINTER
Validates zero unlocalized UI strings, 9-language catalog parity,
and dangling key resolution across client templates and markup.

Rules Enforced:
- Rule 1: Zero hardcoded Vietnamese diacritics in UI code (allow comments & fallback params).
- Rule 2: 100% 9-language dictionary parity for i18n and chat catalogs.
- Rule 3: Resolution of all data-i18n / data-chat-i18n template keys against master catalogs.

Usage:
    python tools/lint/check_i18n_hygiene.py [--strict] [--json-out <path>]
"""

from __future__ import annotations

import argparse
import json
import os
import re
import sys
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Dict, List, Set, Tuple

# Reconfigure stdout/stderr for UTF-8 on Windows
if hasattr(sys.stdout, "reconfigure"):
    sys.stdout.reconfigure(encoding="utf-8")
if hasattr(sys.stderr, "reconfigure"):
    sys.stderr.reconfigure(encoding="utf-8")

PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
CLIENT_DIR = PROJECT_ROOT / "client" / "webapp"
CHAT_UI_JS = CLIENT_DIR / "js" / "ui" / "chat_ui.js"
CHAT_CATALOG_JS = CLIENT_DIR / "js" / "data" / "chat_i18n_catalog.js"
I18N_CATALOG_JS = CLIENT_DIR / "js" / "data" / "i18n_catalog.js"
INDEX_HTML = CLIENT_DIR / "index.html"
TEMPLATES_DIR = CLIENT_DIR / "js" / "ui" / "templates"

ALL_9_LOCALES = ["vi", "en", "zh", "ja", "ko", "th", "de", "ru", "es"]
VI_DIACRITICS_REGEX = re.compile(r"[\u00C0-\u1EF9]")


@dataclass
class LintViolation:
    rule_id: str
    severity: str
    file_path: str
    line_number: int
    message: str
    snippet: str = ""


@dataclass
class LintSummary:
    total_files_scanned: int = 0
    rule1_violations: List[LintViolation] = field(default_factory=list)
    rule2_violations: List[LintViolation] = field(default_factory=list)
    rule3_violations: List[LintViolation] = field(default_factory=list)

    @property
    def total_errors(self) -> int:
        return (
            len(self.rule1_violations)
            + len(self.rule2_violations)
            + len(self.rule3_violations)
        )


def strip_comments_preserve_lines(content: str) -> str:
    """Removes JS single-line and multi-line comments while maintaining line indices."""
    def block_repl(m: re.Match[str]) -> str:
        return "\n" * m.group(0).count("\n")

    cleaned = re.sub(r"/\*[\s\S]*?\*/", block_repl, content)
    return re.sub(r"//.*$", "", cleaned, flags=re.MULTILINE)


def check_rule1_hardcoded_vietnamese(target_file: Path) -> List[LintViolation]:
    """Rule 1: Scans UI JS code for unlocalized Vietnamese strings outside allowlisted structures."""
    violations: List[LintViolation] = []
    if not target_file.exists():
        return violations

    raw_text = target_file.read_text(encoding="utf-8")
    stripped = strip_comments_preserve_lines(raw_text)

    # Mask allowlisted seed dictionary definitions and fallback structures
    masked = re.sub(
        r"export\s+const\s+(?:CHANNELS|RARITY_INFO|ELEMENT_NAMES|CHANNEL_I18N|RARITY_NAMES)\s*=\s*\{[\s\S]*?\};",
        lambda m: "\n" * m.group(0).count("\n"),
        stripped,
    )
    # Mask allowlisted 3rd argument in t('key', params, 'fallback') calls
    masked = re.sub(
        r"\bt\s*\(\s*['\"][^'\"]+['\"]\s*,\s*[^,]+,\s*['\"`][^'\"`]*['\"`]\s*\)",
        "t_allowlisted_call()",
        masked,
    )

    lines = masked.split("\n")
    for line_idx, line in enumerate(lines, start=1):
        for lit_m in re.finditer(r"['\"`]([^'\"`]+)['\"`]", line):
            literal_val = lit_m.group(1)
            if VI_DIACRITICS_REGEX.search(literal_val):
                rel_path = str(target_file.relative_to(PROJECT_ROOT))
                violations.append(
                    LintViolation(
                        rule_id="RULE-1-HARDCODED-VI",
                        severity="ERROR",
                        file_path=rel_path,
                        line_number=line_idx,
                        message=f"Hardcoded Vietnamese string detected: '{literal_val}'",
                        snippet=line.strip()[:100],
                    )
                )
    return violations


def parse_catalog_keys(catalog_path: Path) -> Dict[str, Set[str]]:
    """Extracts localized key sets for all 9 locales from a JavaScript catalog file."""
    if not catalog_path.exists():
        return {loc: set() for loc in ALL_9_LOCALES}

    raw_content = catalog_path.read_text(encoding="utf-8")
    content = strip_comments_preserve_lines(raw_content)
    result: Dict[str, Set[str]] = {}

    for loc in ALL_9_LOCALES:
        m = re.search(rf"\b{loc}\s*:\s*\{{", content)
        if not m:
            continue
        start_idx = m.end()
        braces = 1
        curr = start_idx
        while curr < len(content) and braces > 0:
            if content[curr] == "{":
                braces += 1
            elif content[curr] == "}":
                braces -= 1
            curr += 1

        block = content[start_idx - 1 : curr]
        keys: Set[str] = set()
        entry_pat = re.compile(
            r'(?:[{,]\s*)\s*(?:["\']([a-zA-Z0-9_]+)["\']|([a-zA-Z0-9_]+))\s*:'
        )
        for entry_m in entry_pat.finditer(block):
            k = entry_m.group(1) or entry_m.group(2)
            keys.add(k)
        result[loc] = keys

    return result


def check_rule2_dictionary_parity(catalog_path: Path, catalog_name: str) -> List[LintViolation]:
    """Rule 2: Enforces symmetric 1-to-1 key parity across all 9 supported languages."""
    violations: List[LintViolation] = []
    keys_by_locale = parse_catalog_keys(catalog_path)
    rel_path = str(catalog_path.relative_to(PROJECT_ROOT))

    # Check all 9 locales exist
    for loc in ALL_9_LOCALES:
        if loc not in keys_by_locale or len(keys_by_locale[loc]) == 0:
            violations.append(
                LintViolation(
                    rule_id="RULE-2-DICT-PARITY",
                    severity="ERROR",
                    file_path=rel_path,
                    line_number=1,
                    message=f"{catalog_name}: Missing locale dictionary '{loc}'",
                )
            )

    vi_keys = keys_by_locale.get("vi", set())
    for loc in [l for l in ALL_9_LOCALES if l != "vi"]:
        loc_keys = keys_by_locale.get(loc, set())
        missing_in_loc = vi_keys - loc_keys
        extra_in_loc = loc_keys - vi_keys

        if missing_in_loc:
            sample = sorted(list(missing_in_loc))[:5]
            violations.append(
                LintViolation(
                    rule_id="RULE-2-DICT-PARITY",
                    severity="ERROR",
                    file_path=rel_path,
                    line_number=1,
                    message=f"{catalog_name}: Locale '{loc}' missing {len(missing_in_loc)} keys (e.g. {sample})",
                )
            )
        if extra_in_loc:
            sample = sorted(list(extra_in_loc))[:5]
            violations.append(
                LintViolation(
                    rule_id="RULE-2-DICT-PARITY",
                    severity="ERROR",
                    file_path=rel_path,
                    line_number=1,
                    message=f"{catalog_name}: Locale '{loc}' has {len(extra_in_loc)} extra keys not in 'vi' (e.g. {sample})",
                )
            )
    return violations


def extract_template_referenced_keys(target_dir: Path, index_file: Path) -> List[Tuple[str, str, int]]:
    """Extracts data-i18n, data-chat-i18n, and key attributes from HTML and templates."""
    references: List[Tuple[str, str, int]] = []
    files_to_scan = [index_file]
    if target_dir.exists():
        files_to_scan.extend(sorted(target_dir.glob("*.js")))

    for fpath in files_to_scan:
        if not fpath.exists():
            continue
        rel_path = str(fpath.relative_to(PROJECT_ROOT))
        content = fpath.read_text(encoding="utf-8")
        lines = content.split("\n")
        for line_no, line in enumerate(lines, start=1):
            for m in re.finditer(r"data-(?:chat-)?i18n(?:-[a-z]+)?=[\"']([^\"']+)[\"']", line):
                references.append((m.group(1), rel_path, line_no))
            for m in re.finditer(r"data-(?:tooltip-(?:title|desc)-key|i18n-tooltip-(?:title|desc))=[\"']([^\"']+)[\"']", line):
                references.append((m.group(1), rel_path, line_no))
    return references


def check_rule3_missing_keys(all_defined_keys: Set[str]) -> List[LintViolation]:
    """Rule 3: Detects dangling key references in HTML and templates not present in catalogs."""
    violations: List[LintViolation] = []
    referenced_keys = extract_template_referenced_keys(TEMPLATES_DIR, INDEX_HTML)

    for key_name, fpath, line_no in referenced_keys:
        if key_name not in all_defined_keys:
            violations.append(
                LintViolation(
                    rule_id="RULE-3-DANGLING-KEY",
                    severity="ERROR",
                    file_path=fpath,
                    line_number=line_no,
                    message=f"Referenced i18n key '{key_name}' not defined in any catalog",
                )
            )
    return violations


def run_audit(target_modules: List[Path]) -> LintSummary:
    """Orchestrates all three rules and compiles complete lint summary."""
    summary = LintSummary()
    summary.total_files_scanned = len(target_modules) + 3

    # Rule 1: Scan target UI modules
    for mod in target_modules:
        v1 = check_rule1_hardcoded_vietnamese(mod)
        summary.rule1_violations.extend(v1)

    # Rule 2: Catalog Parity Checks
    v2_chat = check_rule2_dictionary_parity(CHAT_CATALOG_JS, "CHAT_I18N_CATALOG")
    v2_i18n = check_rule2_dictionary_parity(I18N_CATALOG_JS, "I18N_CATALOG")
    summary.rule2_violations.extend(v2_chat)
    summary.rule2_violations.extend(v2_i18n)

    # Rule 3: Missing Key Resolution
    chat_keys = parse_catalog_keys(CHAT_CATALOG_JS).get("vi", set())
    i18n_keys = parse_catalog_keys(I18N_CATALOG_JS).get("vi", set())
    all_defined = chat_keys | i18n_keys
    v3 = check_rule3_missing_keys(all_defined)
    summary.rule3_violations.extend(v3)

    return summary


def print_report(summary: LintSummary, strict: bool) -> None:
    """Formats and prints the human-readable audit report to console."""
    print("=" * 80)
    print("FREEEXILE I18N HYGIENE & ANTI-REGRESSION AUDIT REPORT")
    print("=" * 80)
    print(f"[*] Total Target Files Scanned: {summary.total_files_scanned}")
    print(f"[*] Rule 1 (Zero Hardcoded VI Strings) Violations: {len(summary.rule1_violations)}")
    print(f"[*] Rule 2 (9-Language Parity) Violations        : {len(summary.rule2_violations)}")
    print(f"[*] Rule 3 (Missing / Dangling Keys) Violations  : {len(summary.rule3_violations)}")
    print("-" * 80)

    all_v = (
        summary.rule1_violations
        + summary.rule2_violations
        + summary.rule3_violations
    )
    if all_v:
        for v in all_v:
            print(f"  ❌ [{v.rule_id}] {v.file_path}:{v.line_number} - {v.message}")
            if v.snippet:
                print(f"     Snippet: {v.snippet}")
        print("=" * 80)
        print(f"FAILED: Found {len(all_v)} internationalization hygiene error(s).")
    else:
        print("=" * 80)
        print("✅ SUCCESS: 100% i18n hygiene compliance. All rules passed cleanly!")
    print("=" * 80)


def main() -> int:
    parser = argparse.ArgumentParser(description="FreeExile i18n Hygiene & Parity Static Linter")
    parser.add_argument("--strict", action="store_true", help="Exit code 1 on any violation")
    parser.add_argument("--json-out", type=str, default="", help="Path to write JSON report")
    parser.add_argument(
        "--target-files",
        nargs="*",
        default=[str(CHAT_UI_JS)],
        help="Target UI JavaScript files to scan for Rule 1",
    )
    args = parser.parse_args()

    targets = [Path(p).resolve() for p in args.target_files]
    summary = run_audit(targets)
    print_report(summary, args.strict)

    if args.json_out:
        out_path = Path(args.json_out)
        out_path.parent.mkdir(parents=True, exist_ok=True)
        report_data = {
            "total_files_scanned": summary.total_files_scanned,
            "total_errors": summary.total_errors,
            "rule1": [asdict(v) for v in summary.rule1_violations],
            "rule2": [asdict(v) for v in summary.rule2_violations],
            "rule3": [asdict(v) for v in summary.rule3_violations],
        }
        out_path.write_text(json.dumps(report_data, indent=2, ensure_ascii=False), encoding="utf-8")
        print(f"[*] JSON report written to: {out_path}")

    if args.strict and summary.total_errors > 0:
        return 1
    return 0


if __name__ == "__main__":
    sys.exit(main())
