import sys
import os
import re
import json
import urllib.request
import urllib.error

if hasattr(sys.stdout, 'reconfigure'):
    sys.stdout.reconfigure(encoding='utf-8')

BRIDGE_URL = "http://127.0.0.1:8766"
LANGUAGES = ['vi', 'en', 'de', 'fr', 'es', 'ja']
EPISODES = ['full'] + [str(i) for i in range(1, 9)]

HISTORICAL_MUST_HAVE = [
    "Gia Định",
    "Tây Sơn",
    "Nguyễn Huệ",
    "Chiêu Tăng",
    "Mỹ Tho",
    "Trương Văn Đa",
    "của ta"
]

HISTORICAL_MUST_NOT_HAVE = [
    "gia đình",  # commonly mistranscribed for Gia Định in military contexts
    "của tà",
    "xeam",
    "Mỹ Toa",
    "quý nhân"
]

def test_file_integrity():
    print("=== 1. VERIFYING LOCAL SRT FILE INTEGRITY ===")
    errors = []
    total_checked = 0
    
    for ep in EPISODES:
        dir_name = "full" if ep == "full" else f"ep{ep}"
        for lang in LANGUAGES:
            prefix = "full" if ep == "full" else f"ep{ep}"
            path = f"content/subtitles/{dir_name}/{prefix}_{lang}.srt"
            if not os.path.exists(path):
                errors.append(f"Missing file: {path}")
                continue
            
            size = os.path.getsize(path)
            if size < 100:
                errors.append(f"File too small ({size} bytes): {path}")
                continue
                
            content = open(path, encoding='utf-8').read()
            blocks = re.split(r'\n\s*\n', content.strip())
            if len(blocks) < 5:
                errors.append(f"Too few subtitle cues ({len(blocks)}): {path}")
                continue
                
            # Verify timestamp format
            first_block = blocks[0].strip().splitlines()
            if len(first_block) < 3 or '-->' not in first_block[1]:
                errors.append(f"Malformed SRT header: {path}")
                
            total_checked += 1

    print(f"Checked {total_checked} files. Errors: {len(errors)}")
    assert len(errors) == 0, f"Integrity errors found: {errors}"
    print(">>> 100% of 54 local SRT files are structurally valid!")

def test_historical_accuracy():
    print("\n=== 2. VERIFYING HISTORICAL DIALOGUE ACCURACY (VIETNAMESE) ===")
    vi_full_path = "content/subtitles/full/full_vi.srt"
    content = open(vi_full_path, encoding='utf-8').read()
    
    for term in HISTORICAL_MUST_HAVE:
        count = content.count(term)
        assert count > 0, f"Expected historical term '{term}' not found in {vi_full_path}!"
        print(f"  [PASS] Term '{term}': found {count} occurrences")
        
    for bad_term in HISTORICAL_MUST_NOT_HAVE:
        assert bad_term not in content, f"Found corrupted term '{bad_term}' in {vi_full_path}!"
        print(f"  [PASS] Clean check: '{bad_term}' is NOT present")
        
    print(">>> Historical vocabulary audit passed with 100% precision!")

def test_bridge_api():
    print("\n=== 3. VERIFYING BRIDGE SERVICE API (PORT 8766) ===")
    # 1. Status
    res = urllib.request.urlopen(f"{BRIDGE_URL}/api/status")
    status = json.loads(res.read().decode())
    assert status.get("status") == "online"
    print(f"  [PASS] Service status: {status['status']} (v{status['version']})")
    
    # 2. Subtitles list catalog
    res = urllib.request.urlopen(f"{BRIDGE_URL}/api/subtitles/list")
    data = json.loads(res.read().decode('utf-8'))
    assert data.get("status") == "success"
    episodes = data.get("episodes", {})
    assert len(episodes) == 9
    
    for ep, ep_data in episodes.items():
        for lang in LANGUAGES:
            sub = ep_data["subtitles"].get(lang, {})
            assert sub.get("exists") is True, f"Catalog reports missing {ep}/{lang}"
            assert sub.get("size_bytes", 0) > 100
    print("  [PASS] Catalog /api/subtitles/list: all 9 entities x 6 languages exist and verified")
    
    # 3. Test Text, Download and JSON endpoints
    for lang in LANGUAGES:
        # Text
        res = urllib.request.urlopen(f"{BRIDGE_URL}/api/subtitles/full/{lang}")
        text = res.read().decode('utf-8')
        assert len(text) > 1000
        assert "-->" in text
        
        # Download
        res = urllib.request.urlopen(f"{BRIDGE_URL}/api/subtitles/full/{lang}?download=true")
        assert res.headers.get("content-type") == "application/x-subrip"
        
        # JSON
        res = urllib.request.urlopen(f"{BRIDGE_URL}/api/subtitles/full/{lang}?format=json")
        j = json.loads(res.read().decode('utf-8'))
        assert j.get("status") == "success"
        assert j.get("cue_count", 0) > 50
        assert len(j.get("cues", [])) > 50
        print(f"  [PASS] Language [{lang.upper()}]: PlainText ({len(text)} B), Download (Header OK), JSON ({j['cue_count']} cues)")

    print(">>> Bridge Subtitles API: 100% verified across all modes and languages!")

if __name__ == "__main__":
    test_file_integrity()
    test_historical_accuracy()
    test_bridge_api()
    print("\n=======================================================")
    print("🎉 ALL SUBTITLE VERIFICATION TESTS PASSED SUCCESSFULLY!")
    print("=======================================================")
