"""
Adversarial UI, Microcopy & Typography Stress Test Harness for FreeExile.
Empirically stress-tests:
1. Every interactive control in index.html and all template_catalog_*.js has word count <= 2 words.
2. Zero parenthetical substrings '(' or ')' in user-facing UI labels and headings.
3. Zero serif font references (Cinzel, Noto Serif, Georgia, serif) in all CSS and JS renderers.
4. Executes standard test suites.
"""

from __future__ import annotations
import os
import re
import sys
import json
import subprocess
from pathlib import Path
from typing import Dict, List, Set, Tuple, Any
from bs4 import BeautifulSoup

if hasattr(sys.stdout, "reconfigure"):
    sys.stdout.reconfigure(encoding="utf-8")
if hasattr(sys.stderr, "reconfigure"):
    sys.stderr.reconfigure(encoding="utf-8")

PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
WEBAPP_DIR = PROJECT_ROOT / "client" / "webapp"
INDEX_HTML = WEBAPP_DIR / "index.html"
CSS_DIR = WEBAPP_DIR / "css"
TEMPLATES_DIR = WEBAPP_DIR / "js" / "ui" / "templates"
ENGINE_DIR = WEBAPP_DIR / "js" / "engine"
UI_DIR = WEBAPP_DIR / "js" / "ui"
I18N_CATALOG_JS = WEBAPP_DIR / "js" / "data" / "i18n_catalog.js"

ALL_9_LOCALES = ["vi", "en", "zh", "ja", "ko", "th", "de", "ru", "es"]

# Regex for emojis & symbols
EMOJI_PATTERN = re.compile(
    r"[\U00010000-\U0010ffff\u2600-\u27ff\ufe00-\ufe0f\u200d\u2300-\u23ff\u2b50\u2b55\u25a0-\u25ff"
    r"⚔️🗺️🏕️✕🪵🐺👹⚡🩸🛡️🧹🌀💭💰💎🗝️💀🗡️✓•]"
)

# Serif detection regex: Matches 'serif' not preceded by 'sans-', 'Noto Serif', 'Cinzel', 'Georgia'
SERIF_REGEX = re.compile(r"(?<!sans-)serif|Noto\s*Serif|Cinzel|Georgia", re.IGNORECASE)


def extract_words(text: str) -> List[str]:
    """Adversarial word extraction handling whitespace, html entities, punctuation, emojis."""
    t = text.replace("&nbsp;", " ").replace("&#160;", " ")
    # remove inline tags
    t = re.sub(r"<[^>]+>", " ", t)
    # remove emojis
    t = EMOJI_PATTERN.sub(" ", t)
    # remove punctuation
    t = re.sub(r"[\(\)\[\]\{\}\:\|\,\.\!\?\/\-\+\—\–\<\>\"\'`\*\#\@\$\%\^\&\~]", " ", t)
    # remove symbols, keep unicode letters and numbers
    t = re.sub(r"[^\w\s]", " ", t)
    words = [w for w in t.split() if w and not w.isdigit()]
    return words


def parse_template_js_strings(content: str) -> List[Tuple[str, str]]:
    """Extract registered template ids and their raw HTML string arguments."""
    templates = []
    # Match reg('modal-id', `...`) or reg("modal-id", "...")
    matches = re.finditer(r"reg\s*\(\s*['\"]([^'\"]+)['\"]\s*,\s*([`'\"])(.*?)\2\s*\)", content, re.DOTALL)
    for m in matches:
        tid = m.group(1)
        html_str = m.group(3)
        templates.append((tid, html_str))
    return templates


def load_i18n_catalog() -> Dict[str, Dict[str, str]]:
    """Load I18N_CATALOG via node.js."""
    escaped_path = str(I18N_CATALOG_JS).replace("\\", "/")
    node_script = f"""
    const fs = require('fs');
    const code = fs.readFileSync('{escaped_path}', 'utf-8');
    const fakeWindow = {{}};
    new Function('window', code)(fakeWindow);
    console.log(JSON.stringify(fakeWindow.I18N_CATALOG));
    """
    res = subprocess.run(["node", "-e", node_script], capture_output=True, encoding="utf-8", check=True)
    return json.loads(res.stdout)


def test_interactive_controls_word_count() -> List[Dict[str, Any]]:
    """Test 1: Check that all interactive controls have <= 2 words (static and localized across 9 languages)."""
    violations = []
    catalog = load_i18n_catalog()
    
    def check_element(el, source: str):
        tag_name = el.name
        el_id = el.get("id", "no-id")
        el_class = " ".join(el.get("class", []))
        
        # If input, only check buttons/submits
        if tag_name == "input":
            itype = el.get("type", "text")
            if itype not in ["button", "submit", "reset"]:
                return
            text = el.get("value", "")
        else:
            text = el.get_text(separator=" ", strip=True)
            
        words = extract_words(text)
        if len(words) > 2:
            violations.append({
                "source": source,
                "tag": tag_name,
                "id": el_id,
                "class": el_class,
                "text": text,
                "words": words,
                "word_count": len(words)
            })
            
        # Check data-i18n resolution across all 9 languages
        i18n_key = el.get("data-i18n")
        if i18n_key:
            for loc in ALL_9_LOCALES:
                loc_val = catalog.get(loc, {}).get(i18n_key, "")
                if loc_val:
                    if loc in ["vi", "en", "de", "es", "ko", "ru"]:
                        lwords = extract_words(loc_val)
                        if len(lwords) > 2:
                            violations.append({
                                "source": f"{source} (i18n [{loc}])",
                                "tag": tag_name,
                                "id": el_id,
                                "i18n_key": i18n_key,
                                "text": loc_val,
                                "words": lwords,
                                "word_count": len(lwords)
                            })
                    elif loc == "zh":
                        clean_c = [c for c in loc_val if not c.isspace() and not EMOJI_PATTERN.search(c)]
                        if len(clean_c) > 4:
                            violations.append({
                                "source": f"{source} (i18n [{loc}])",
                                "tag": tag_name,
                                "id": el_id,
                                "i18n_key": i18n_key,
                                "text": loc_val,
                                "word_count": len(clean_c),
                                "reason": f"CJK exceeds 4 chars ({len(clean_c)})"
                            })
                    elif loc == "ja":
                        clean_c = [c for c in loc_val if not c.isspace() and not EMOJI_PATTERN.search(c)]
                        if len(clean_c) > 6:
                            violations.append({
                                "source": f"{source} (i18n [{loc}])",
                                "tag": tag_name,
                                "id": el_id,
                                "i18n_key": i18n_key,
                                "text": loc_val,
                                "word_count": len(clean_c),
                                "reason": f"JA exceeds 6 chars ({len(clean_c)})"
                            })
                    elif loc == "th":
                        clean_th = EMOJI_PATTERN.sub("", loc_val).strip()
                        if len(clean_th) > 12:
                            violations.append({
                                "source": f"{source} (i18n [{loc}])",
                                "tag": tag_name,
                                "id": el_id,
                                "i18n_key": i18n_key,
                                "text": loc_val,
                                "word_count": len(clean_th),
                                "reason": f"TH exceeds 12 chars ({len(clean_th)})"
                            })

    # 1. Parse index.html
    html_content = INDEX_HTML.read_text(encoding="utf-8")
    soup_html = BeautifulSoup(html_content, "html.parser")
    
    interactive_elements = soup_html.find_all(
        lambda el: el.name in ["button", "a", "input"] or el.get("role") in ["button", "tab"] or any("tab" in c for c in el.get("class", []))
    )
    for el in interactive_elements:
        check_element(el, "index.html")

    # 2. Parse all template catalogs
    for tpl_file in TEMPLATES_DIR.glob("*.js"):
        content = tpl_file.read_text(encoding="utf-8")
        templates = parse_template_js_strings(content)
        for tid, tpl_html in templates:
            tsoup = BeautifulSoup(tpl_html, "html.parser")
            t_interactive = tsoup.find_all(
                lambda el: el.name in ["button", "a", "input"] or el.get("role") in ["button", "tab"] or any("tab" in c for c in el.get("class", []))
            )
            for el in t_interactive:
                check_element(el, f"{tpl_file.name} [{tid}]")
                    
    return violations


def test_zero_parenthetical_labels_and_headings() -> List[Dict[str, Any]]:
    """Test 2: Check for any '(' or ')' in user-facing UI labels, headings, tooltips."""
    violations = []
    
    # 1. Check index.html
    html_content = INDEX_HTML.read_text(encoding="utf-8")
    soup_html = BeautifulSoup(html_content, "html.parser")
    
    # Check headings h1-h6, labels, buttons, a
    for el in soup_html.find_all(["h1", "h2", "h3", "h4", "h5", "h6", "label", "button", "a"]):
        text = el.get_text(strip=True)
        if "(" in text or ")" in text:
            violations.append({
                "source": "index.html",
                "tag": el.name,
                "id": el.get("id", "no-id"),
                "text": text,
                "reason": "Text contains '(' or ')'"
            })
            
    # Check tooltip attributes in index.html
    for el in soup_html.find_all(True):
        for attr in ["data-tooltip-title", "title", "aria-label", "placeholder"]:
            val = el.get(attr)
            if val and ("(" in val or ")" in val):
                violations.append({
                    "source": "index.html",
                    "tag": el.name,
                    "id": el.get("id", "no-id"),
                    "attr": attr,
                    "value": val,
                    "reason": f"Attribute {attr} contains '(' or ')'"
                })

    # 2. Check all templates
    for tpl_file in TEMPLATES_DIR.glob("*.js"):
        content = tpl_file.read_text(encoding="utf-8")
        templates = parse_template_js_strings(content)
        for tid, tpl_html in templates:
            tsoup = BeautifulSoup(tpl_html, "html.parser")
            for el in tsoup.find_all(["h1", "h2", "h3", "h4", "h5", "h6", "label", "button", "a"]):
                text = el.get_text(strip=True)
                if "(" in text or ")" in text:
                    violations.append({
                        "source": f"{tpl_file.name} [{tid}]",
                        "tag": el.name,
                        "id": el.get("id", "no-id"),
                        "text": text,
                        "reason": "Template text contains '(' or ')'"
                    })
            for el in tsoup.find_all(True):
                for attr in ["data-tooltip-title", "title", "aria-label", "placeholder"]:
                    val = el.get(attr)
                    if val and ("(" in val or ")" in val):
                        violations.append({
                            "source": f"{tpl_file.name} [{tid}]",
                            "tag": el.name,
                            "id": el.get("id", "no-id"),
                            "attr": attr,
                            "value": val,
                            "reason": f"Template attribute {attr} contains '(' or ')'"
                        })

    # 3. Check i18n_catalog.js UI labels / titles / buttons
    catalog = load_i18n_catalog()
    for loc, d in catalog.items():
        for k, v in d.items():
            # Check UI labels/buttons/tabs/titles
            if not k.endswith("_desc") and not k.endswith("_subtitle") and not k.endswith("_content"):
                if "(" in v or ")" in v:
                    violations.append({
                        "source": f"i18n_catalog.js [{loc}]",
                        "key": k,
                        "value": v,
                        "reason": f"i18n key '{k}' in [{loc}] contains '(' or ')'"
                    })

    return violations


def test_zero_serif_font_declarations() -> List[Dict[str, Any]]:
    """Test 3: Scan all CSS files and JS renderers for serif font declarations."""
    violations = []
    
    # 1. Scan all CSS files
    for css_file in CSS_DIR.glob("*.css"):
        lines = css_file.read_text(encoding="utf-8").splitlines()
        for idx, line in enumerate(lines, start=1):
            if SERIF_REGEX.search(line):
                violations.append({
                    "file": str(css_file.relative_to(PROJECT_ROOT)),
                    "line_number": idx,
                    "line_content": line.strip(),
                    "reason": "Serif font detected in CSS"
                })

    # 2. Scan JS renderers
    renderer_files = [
        ENGINE_DIR / "world_renderer.js",
        ENGINE_DIR / "vfx_renderer.js",
        ENGINE_DIR / "canvas_renderer.js",
        ENGINE_DIR / "entity_renderer.js",
        ENGINE_DIR / "telegraph_renderer.js",
        ENGINE_DIR / "weapon_swing_renderer.js",
        UI_DIR / "endgame_atlas.js",
        UI_DIR / "war_fog_renderer.js",
        UI_DIR / "hud_orbs.js",
        UI_DIR / "chat_ui.js"
    ]
    
    for rf in renderer_files:
        if not rf.exists():
            continue
        lines = rf.read_text(encoding="utf-8").splitlines()
        for idx, line in enumerate(lines, start=1):
            if SERIF_REGEX.search(line):
                violations.append({
                    "file": str(rf.relative_to(PROJECT_ROOT)),
                    "line_number": idx,
                    "line_content": line.strip(),
                    "reason": "Serif font detected in JS renderer"
                })

    # 3. Scan index.html (Tailwind config, font links, inline styles)
    index_lines = INDEX_HTML.read_text(encoding="utf-8").splitlines()
    for idx, line in enumerate(index_lines, start=1):
        if SERIF_REGEX.search(line):
            violations.append({
                "file": "client/webapp/index.html",
                "line_number": idx,
                "line_content": line.strip(),
                "reason": "Serif font detected in index.html"
            })

    return violations


def run_standard_suites() -> Dict[str, Any]:
    """Test 4: Execute pytest and unittest suites."""
    results = {}
    
    # Pytest E2E
    cmd1 = ["pytest", "tests/e2e/test_ui_typography_i18n_wiki_streamlining_e2e.py", "-v"]
    p1 = subprocess.run(cmd1, cwd=PROJECT_ROOT, capture_output=True, encoding="utf-8", errors="replace")
    results["pytest_e2e"] = {
        "exit_code": p1.returncode,
        "stdout": p1.stdout,
        "stderr": p1.stderr,
        "passed": p1.returncode == 0
    }
    
    # Unittest Localization Engine
    cmd2 = ["python", "-m", "unittest", "tests/unit/test_webapp_localization_engine.py", "-v"]
    p2 = subprocess.run(cmd2, cwd=PROJECT_ROOT, capture_output=True, encoding="utf-8", errors="replace")
    results["unittest_localization"] = {
        "exit_code": p2.returncode,
        "stdout": p2.stdout,
        "stderr": p2.stderr,
        "passed": p2.returncode == 0
    }
    
    return results


def main():
    print("=" * 70)
    print("EMPIRICAL ADVERSARIAL STRESS TEST HARNESS — CHALLENGER STREAMLINE 1")
    print("=" * 70)
    
    print("\n--- 1. Testing Interactive Controls Word Count (<= 2 Words) ---")
    wc_violations = test_interactive_controls_word_count()
    print(f"Total Interactive Word Count Violations: {len(wc_violations)}")
    for v in wc_violations[:10]:
        details = v.get('words', v.get('reason', ''))
        print(f"  [FAIL] {v['source']} <{v['tag']} id='{v['id']}'>: text='{v['text']}' -> {details} ({v['word_count']} words/chars)")
    if len(wc_violations) > 10:
        print(f"  ... and {len(wc_violations) - 10} more.")
        
    print("\n--- 2. Testing Parenthetical Substrings '(' or ')' in UI Labels ---")
    paren_violations = test_zero_parenthetical_labels_and_headings()
    print(f"Total Parenthetical Violations: {len(paren_violations)}")
    for v in paren_violations[:10]:
        print(f"  [FAIL] {v['source']}: {v.get('text') or v.get('value')} ({v['reason']})")
    if len(paren_violations) > 10:
        print(f"  ... and {len(paren_violations) - 10} more.")

    print("\n--- 3. Testing Zero Serif Font References in CSS & JS Renderers ---")
    serif_violations = test_zero_serif_font_declarations()
    print(f"Total Serif Font Violations: {len(serif_violations)}")
    for v in serif_violations:
        print(f"  [FAIL] {v['file']}:{v['line_number']} -> {v['line_content']}")

    print("\n--- 4. Running Standard Test Suites ---")
    suites = run_standard_suites()
    print(f"pytest E2E: {'PASS' if suites['pytest_e2e']['passed'] else 'FAIL'} (exit {suites['pytest_e2e']['exit_code']})")
    print(f"unittest Localization: {'PASS' if suites['unittest_localization']['passed'] else 'FAIL'} (exit {suites['unittest_localization']['exit_code']})")

    all_passed = (
        len(wc_violations) == 0 and
        len(paren_violations) == 0 and
        len(serif_violations) == 0 and
        suites["pytest_e2e"]["passed"] and
        suites["unittest_localization"]["passed"]
    )
    
    verdict = "APPROVE" if all_passed else "REQUEST_CHANGES"
    print("\n" + "=" * 70)
    print(f"FINAL ADVERSARIAL VERDICT: {verdict}")
    print("=" * 70)
    
    # Dump full json report for analysis
    report_data = {
        "verdict": verdict,
        "interactive_word_count_violations": wc_violations,
        "parenthetical_violations": paren_violations,
        "serif_font_violations": serif_violations,
        "test_suites": {
            "pytest_e2e": {
                "exit_code": suites["pytest_e2e"]["exit_code"],
                "passed": suites["pytest_e2e"]["passed"]
            },
            "unittest_localization": {
                "exit_code": suites["unittest_localization"]["exit_code"],
                "passed": suites["unittest_localization"]["passed"]
            }
        }
    }
    
    report_path = Path(__file__).resolve().parent / "adversarial_results.json"
    report_path.write_text(json.dumps(report_data, indent=2, ensure_ascii=False), encoding="utf-8")
    print(f"Detailed results saved to: {report_path}")
    
    sys.exit(0 if all_passed else 1)


if __name__ == "__main__":
    main()
