# -*- coding: utf-8 -*-
"""
Unit test suite for FreeExile 9-Language Novel Localization Integrity.
Validates 100% adherence to canonical nomenclature, 216 chapter files,
0 residual Vietnamese characters in non-VI editions, and 0 vulgarity/anomalies.
"""

import os
import re
import glob
import pytest
from pathlib import Path

from tools.localization.novel_titles_catalog import CHAPTER_TITLES
from tools.localization.novel_headings_catalog import CANONICAL_SECTION_HEADINGS
from tools.localization.novel_entities_catalog import ENTITY_DICTIONARY

LANGUAGES = ["vi", "en", "zh", "ja", "ko", "th", "de", "ru", "es"]
NON_VI_LANGUAGES = ["en", "zh", "ja", "ko", "th", "de", "ru", "es"]

PURE_VI_DIACRITICS = re.compile(r"[ăâđêôơưĂÂĐÊÔƠƯằầềồờừẳẩẻểỉỏổởủửỷẵẫẽễĩõỗỡũữỹạặậẹệịọộợụựỵ]")

ANOMALOUS_STRINGS = [
    "Con Cot", "Nuoc Duc", "Hac Thi", "Hoang Cot Tho", "Hung Sa Cot",
    "Bang Loi", "Lai Bo Thi Lang", "Dai Doanh", "Co Thi",
    "Kim Van Nhuc", "Vu Hoa Shinh", "floorable sandwich bowl",
    "Menstrual Hate", "SEXUALITY RISE", "按摩黑人", "KIM LOAN ELECTRICITY",
    "TWO BINH CHANGE", "Chan Thuy", "Chan Thien", "Thanh Lac", "Hai Mon",
    "Trong Dao", "Qing Lạc", "Jiang Ky", "NIVELA EL JIANG KY", "고위 감귤", "高位のみかん"
]

FORBIDDEN_KINH_KY_PATTERNS = [
    re.compile(r"menstru", re.IGNORECASE),
    re.compile(r"monatsblut", re.IGNORECASE),
    re.compile(r"ประจำเดือน"),
    re.compile(r"经血"),
    re.compile(r"经期"),
    re.compile(r"経血"),
    re.compile(r"経期"),
    re.compile(r"생리"),
    re.compile(r"생리혈"),
    re.compile(r"\bperiods?\b", re.IGNORECASE),
    re.compile(r"\bper[ií]odos?\b", re.IGNORECASE),
    re.compile(r"\bperiode[n]?\b", re.IGNORECASE),
    re.compile(r"\bпериод\w*", re.IGNORECASE),
    re.compile(r"jiang ky", re.IGNORECASE),
    re.compile(r"高位のみかん"),
    re.compile(r"고위 감귤"),
    re.compile(r"capital of death", re.IGNORECASE),
    re.compile(r"死刑期"),
    re.compile(r"死刑時代"),
    re.compile(r"사형시대"),
    re.compile(r"사형수\s*감옥\s*기간"),
    re.compile(r"ช่วงจำคุกประหารชีวิต"),
    re.compile(r"ช่วงแห่งความตาย"),
    re.compile(r"Zeit des Todes", re.IGNORECASE),
    re.compile(r"剑派史诗级"),
    re.compile(r"壮大な時代"),
    re.compile(r"검파의 서사시"),
    re.compile(r"ยุคมหากาพย์"),
    re.compile(r"ゼント\d*"),
    re.compile(r"七年前七年前"),
    re.compile(r"vor sieben Jahren.*vor sieben Jahren"),
]

def test_exactly_216_localized_chapters_exist():
    """Verify that all 24 chapters across 9 canonical languages exist (216 files)."""
    for lang in LANGUAGES:
        for ch in range(1, 25):
            pattern = f"novels/locales/{lang}/act_*/chuong_{ch:03d}_*.md"
            matches = glob.glob(pattern)
            assert len(matches) == 1, f"Missing chapter {ch} for language {lang}"

def test_chapter_titles_match_canonical():
    """Verify that every chapter title level-1 heading matches the canonical catalog."""
    prefix_map = {
        "en": "CHAPTER",
        "de": "KAPITEL",
        "ru": "ГЛАВА",
        "es": "CAPÍTULO",
    }
    for ch in range(1, 25):
        for lang in LANGUAGES:
            files = glob.glob(f"novels/locales/{lang}/act_*/chuong_{ch:03d}_*.md")
            assert files, f"Missing chapter {ch} for {lang}"
            content = Path(files[0]).read_text(encoding="utf-8")
            first_line = content.splitlines()[0].strip()
            title = CHAPTER_TITLES[ch][lang]

            if lang == "vi":
                assert f"CHƯƠNG {ch}: {title}" in first_line
            elif lang in ("zh", "ja"):
                assert f"# 第{ch}章：{title}" == first_line
            elif lang == "ko":
                assert f"# 제{ch}장: {title}" == first_line
            elif lang == "th":
                assert f"# บทที่ {ch}: {title}" == first_line
            else:
                prefix = prefix_map[lang]
                assert f"# {prefix} {ch}: {title}" == first_line

def test_all_113_section_headings_match_canonical():
    """Verify that every section heading in all 216 files matches the canonical headings."""
    for ch in range(1, 25):
        vi_files = glob.glob(f"novels/locales/vi/act_*/chuong_{ch:03d}_*.md")
        vi_content = Path(vi_files[0]).read_text(encoding="utf-8")
        vi_headings = [l.strip() for l in vi_content.splitlines() if l.strip().startswith("### ")]

        for lang in NON_VI_LANGUAGES:
            lang_files = glob.glob(f"novels/locales/{lang}/act_*/chuong_{ch:03d}_*.md")
            lang_content = Path(lang_files[0]).read_text(encoding="utf-8")
            lang_headings = [l.strip() for l in lang_content.splitlines() if l.strip().startswith("### ")]

            assert len(lang_headings) == len(vi_headings), (
                f"Chapter {ch} ({lang}) has {len(lang_headings)} headings, expected {len(vi_headings)}"
            )

            for idx, h in enumerate(lang_headings):
                expected = CANONICAL_SECTION_HEADINGS[vi_headings[idx]][lang]
                assert h == expected, (
                    f"Chapter {ch} ({lang}) heading mismatch: '{h}' != '{expected}'"
                )

def test_zero_pure_vietnamese_diacritics_in_non_vi():
    """Verify that zero pure Vietnamese characters remain in any non-VI edition."""
    violations = []
    for lang in NON_VI_LANGUAGES:
        for fpath in glob.glob(f"novels/locales/{lang}/act_*/*.md"):
            content = Path(fpath).read_text(encoding="utf-8")
            matches = PURE_VI_DIACRITICS.findall(content)
            if matches:
                violations.append((fpath, set(matches)))

    assert not violations, f"Found residual Vietnamese diacritics: {violations}"

def test_zero_anomalous_or_vulgar_strings():
    """Verify that no corrupted tokens or vulgar machine-translation artifacts exist."""
    found = []
    for lang in NON_VI_LANGUAGES:
        for fpath in glob.glob(f"novels/locales/{lang}/act_*/*.md"):
            content = Path(fpath).read_text(encoding="utf-8")
            for token in ANOMALOUS_STRINGS:
                if token in content:
                    found.append((fpath, token))

    assert not found, f"Found anomalous or vulgar tokens: {found}"

def test_zero_menstrual_or_inappropriate_period_terms():
    """Verify that zero menstrual or inappropriate period/era mistranslations of 'kinh kỳ' exist."""
    found = []
    for lang in NON_VI_LANGUAGES:
        for fpath in glob.glob(f"novels/locales/{lang}/act_*/*.md"):
            content = Path(fpath).read_text(encoding="utf-8")
            for pat in FORBIDDEN_KINH_KY_PATTERNS:
                matches = pat.findall(content)
                if matches:
                    found.append((fpath, pat.pattern, matches))

    assert not found, f"Found forbidden menstrual/period mistranslation tokens: {found}"

def test_no_crlf_line_endings_in_any_locale():
    """Verify that all 216 chapter files strictly use LF line endings."""
    crlf_files = []
    for lang in LANGUAGES:
        for fpath in glob.glob(f"novels/locales/{lang}/act_*/*.md"):
            data = Path(fpath).read_bytes()
            if b"\r\n" in data:
                crlf_files.append(fpath)

    assert not crlf_files, f"Files with CRLF endings: {crlf_files}"

def test_catalogs_and_entity_integrity():
    """Verify catalog imports and size limits."""
    assert len(CHAPTER_TITLES) == 24
    assert len(CANONICAL_SECTION_HEADINGS) == 113
    assert len(ENTITY_DICTIONARY) >= 100

    # Ensure all entities are sorted descending by length of VI term
    lengths = [len(e[0]) for e in ENTITY_DICTIONARY]
    assert lengths == sorted(lengths, reverse=True), "ENTITY_DICTIONARY not sorted descending by length"

def test_zero_pov_slips_in_narration():
    """Verify that audited first-person slips in third-person narration are eradicated."""
    checks = [
        ("en", "chuong_001_*.md", "tearing through my dry lungs"),
        ("en", "chuong_009_*.md", "receded to my knees, then to my ankles"),
        ("en", "chuong_010_*.md", "My blade was made of Crude Flint"),
        ("en", "chuong_010_*.md", "In just one slash, I cut"),
        ("en", "chuong_015_*.md", "corners of my dry, cracked lips"),
        ("th", "chuong_010_*.md", "ใบมีดของฉันทำจาก หินเหล็กไฟหยาบ"),
        ("th", "chuong_010_*.md", "ฉันตัดโซ่เหล็กหนา"),
        ("ja", "chuong_022_*.md", "もし私たちがその腐った王座を破壊しなければ"),
    ]
    for lang, glob_pat, forbidden_str in checks:
        for fpath in glob.glob(f"novels/locales/{lang}/act_*/{glob_pat}"):
            content = Path(fpath).read_text(encoding="utf-8")
            assert forbidden_str not in content, f"Found POV slip in {fpath}: '{forbidden_str}'"
