import re
import pathlib
import sys

if hasattr(sys.stdout, "reconfigure"):
    sys.stdout.reconfigure(encoding="utf-8")
    sys.stderr.reconfigure(encoding="utf-8")

root = pathlib.Path(r"c:\Projects\FreeExile\client\webapp")

# Check specifically:
# 1. index.html
# 2. template_catalog_*.js
# 3. i18n_catalog.js
# 4. Other JS files in client/webapp/js/ui/

files_to_check = [root / "index.html"] + list((root / "js" / "ui").rglob("*.js")) + list((root / "js" / "data").rglob("*.js"))

bilingual_findings = []

for f in files_to_check:
    if f.name.endswith(".bak"):
        continue
    content = f.read_text(encoding="utf-8")
    
    # 1. Check for buttons with parens
    buttons = re.findall(r'<button\b[^>]*>(.*?)</button>', content, re.DOTALL)
    for b in buttons:
        # Strip HTML tags
        b_clean = re.sub(r'<[^>]+>', ' ', b).strip()
        m = re.findall(r'\(([^\)]+)\)', b_clean)
        if m:
            bilingual_findings.append({
                "file": f.relative_to(root.parent.parent),
                "type": "button_parens",
                "text": b_clean,
                "parens": m
            })
            
    # 2. Check for tooltip titles or attrs with parens
    for attr in ["data-tooltip-title", "title", "data-tooltip-desc"]:
        vals = re.findall(rf'{attr}=["\']([^"\']+)["\']', content)
        for v in vals:
            m = re.findall(r'\(([^\)]+)\)', v)
            if m:
                # ignore hotkeys like (C), (H), (Space)
                # But check for bilingual strings like (Hideout), (World Map), etc.
                bilingual_findings.append({
                    "file": f.relative_to(root.parent.parent),
                    "type": f"{attr}_parens",
                    "text": v,
                    "parens": m
                })

    # 3. Check for bilingual pattern in text content: e.g. Vietnamese words followed by (English words)
    # e.g., "Động Thiên (Hideout)", "Bản Đồ (World Map)"
    bilingual_patterns = re.findall(r'([\w\sàáảãạăắằẳẵặâấầẩẫậèéẻẽẹêếềểễệìíỉĩịòóỏõọôốồổỗộơớờởỡợùúủũụưứừửữựỳýỷỹỵđÀÁẢÃẠĂẮẰẲẴẶÂẤẦẨẪẬÈÉẺẼẸÊẾỀỂỄỆÌÍỈĨỊÒÓỎÕỌÔỐỒỔỖỘƠỚỜỞỠỢÙÚỦŨỤƯỨỪỬỮỰỲÝỶỸỴĐ]+)\(([A-Za-z\s]+)\)', content)
    for vi, en in bilingual_patterns:
        vi_s = vi.strip()
        en_s = en.strip()
        if len(vi_s.split()) >= 1 and len(en_s.split()) >= 1:
            # ignore short codes or single letters like F11, C, etc.
            if len(en_s) > 3 and en_s not in ("Space", "Shift", "Enter", "Right", "Left"):
                bilingual_findings.append({
                    "file": f.relative_to(root.parent.parent),
                    "type": "bilingual_text_pattern",
                    "text": f"{vi_s} ({en_s})",
                    "parens": [en_s]
                })

print(f"Total bilingual findings: {len(bilingual_findings)}")
for b in bilingual_findings:
    print(f"[{b['file']}] {b['type']}: {b['text']} -> {b['parens']}")
