import json
import re
import sys
from pathlib import Path
import subprocess

sys.stdout.reconfigure(encoding='utf-8')

root = Path(r"c:\Projects\FreeExile")
catalog_file = root / "client" / "webapp" / "js" / "data" / "i18n_catalog.js"
escaped_path = str(catalog_file).replace("\\", "/")

node_script = """
const fs = require('fs');
const code = fs.readFileSync('""" + escaped_path + """', 'utf-8');
const fakeWindow = {};
const fn = new Function('window', code + '; return I18N_CATALOG;');
const catalog = fn(fakeWindow);
console.log(JSON.stringify(catalog));
"""

res = subprocess.run(["node", "-e", node_script], capture_output=True, encoding="utf-8", check=True)
catalog = json.loads(res.stdout)

locales = ["vi", "en", "zh", "ja", "ko", "th", "de", "ru", "es"]
all_keys = set(k for loc in catalog.values() for k in loc.keys())

print("=== 1. FULL KEY PARITY CHECK ===")
parity_failures = {}
for loc in locales:
    missing = all_keys - set(catalog.get(loc, {}).keys())
    if missing:
        parity_failures[loc] = len(missing)
print(f"Parity failures by locale: {parity_failures}")

print("\n=== 2. ZERO BILINGUAL / PARENTHETICAL PATTERNS CHECK ===")
# Exclude templates like {name} or [{count}/6]
paren_regex = re.compile(r"\([^{}]*?\)")
bilingual_failures = []
for loc in locales:
    for k, v in catalog.get(loc, {}).items():
        if isinstance(v, str) and paren_regex.search(v):
            bilingual_failures.append((loc, k, v))
print(f"Total bilingual / parenthetical failures: {len(bilingual_failures)}")
for loc, k, v in bilingual_failures[:10]:
    print(f"   [{loc}] {k}: {v}")

print("\n=== 3. 1-2 WORD MICROCOPY CONSTRAINT CHECK ===")
# Keys that must strictly be 1-2 words:
microcopy_key_prefixes = (
    "npc_", "char_name", "modal_", "vault_tab_", "shop_tab_",
    "dodge", "buyout", "agent_btn_revoke", "agent_stance_", "telemetry_agent_orb",
    "interaction_step_through_portal", "interaction_dialogue", "dialogue_btn_", "dialogue_select"
)
# Exclude dialogue body, dialogue choices, notices, and tooltips desc
microcopy_failures = []
for loc in locales:
    for k, v in catalog.get(loc, {}).items():
        # Check if key is a short UI element
        is_short_ui = False
        if k in ["char_name", "dodge", "buyout", "agent_btn_revoke", "telemetry_agent_orb",
                 "interaction_step_through_portal", "interaction_dialogue", "dialogue_select"]:
            is_short_ui = True
        elif k.startswith("npc_") and (k.endswith("_name") or k.endswith("_title") or k.endswith("_service")):
            is_short_ui = True
        elif k.startswith("modal_") and not k.endswith("_subtitle"):
            is_short_ui = True
        elif k.startswith("vault_tab_") or k.startswith("shop_tab_"):
            is_short_ui = True
        elif k.startswith("agent_stance_"):
            is_short_ui = True
        elif k.startswith("tooltip_") and k.endswith("_title") and k != "char_status_tooltip_title":
            is_short_ui = True
        elif k.startswith("skill_") and not k.startswith("tooltip_"):
            is_short_ui = True

        if is_short_ui:
            # strip emojis
            clean_v = re.sub(r"[\U00010000-\U0010ffff\u2600-\u27ff]", "", v).strip()
            # check length
            if loc in ["vi", "en", "de", "ru", "es", "ko"]:
                words = clean_v.split()
                if len(words) > 2:
                    microcopy_failures.append((loc, k, v, len(words)))
            elif loc in ["zh", "ja"]:
                chars = len([c for c in clean_v if not c.isspace()])
                if chars > 4:
                    microcopy_failures.append((loc, k, v, chars))
            elif loc == "th":
                chars = len(clean_v)
                if chars > 12:
                    microcopy_failures.append((loc, k, v, chars))

print(f"Total microcopy brevity failures: {len(microcopy_failures)}")
for loc, k, v, cnt in microcopy_failures[:15]:
    print(f"   [{loc}] {k}: '{v}' ({cnt} units)")

print("\n=== 4. INDEX.HTML AND TEMPLATES KEY CHECK ===")
webapp_dir = root / "client" / "webapp"
referenced_keys = set()
for p in webapp_dir.rglob("*.html"):
    text = p.read_text(encoding="utf-8")
    for m in re.finditer(r'data-(?:i18n|tooltip-title-key|tooltip-desc-key)=["\']([a-zA-Z0-9_-]+)["\']', text):
        referenced_keys.add(m.group(1))

missing_refs = referenced_keys - all_keys
print(f"Total HTML referenced keys: {len(referenced_keys)}")
print(f"Missing referenced keys from catalog: {missing_refs}")

