import os
import re
import sys
import json

# Ensure UTF-8 output on Windows
sys.stdout.reconfigure(encoding='utf-8')

def run_adversarial_c05_c08_tests():
    base_dir = os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))
    tap01_path = os.path.join(base_dir, 'FilmMaker', 'TAP_01_XUAN_SAC_THE_NGUYEN_VA_GIONG_BAO_DOAN_TRUONG.md')
    idx_path = os.path.join(base_dir, 'FilmMaker', 'INDEX_VA_DANH_MUC_CANH_QUAY.md')

    with open(tap01_path, 'r', encoding='utf-8') as f:
        tap01 = f.read()
    with open(idx_path, 'r', encoding='utf-8') as f:
        idx = f.read()

    test_results = {
        "suite": "M1 Screenplay Challenger 2: Đạm Tiên 3-Beat Sequence & Lip-Sync Guard",
        "timestamp": "2026-10-08T05:10:00Z",
        "tests": {},
        "failures": [],
        "warnings": []
    }

    print("=====================================================================")
    print("ADVERSARIAL STRESS TEST: ĐẠM TIÊN 3-BEAT SEQUENCE (C05) & LIP-SYNC (C08)")
    print("=====================================================================\n")

    # =========================================================================
    # 1. SCENE 05 OVERALL BOUNDARY & SHOT EXTRACTION
    # =========================================================================
    print("--- TEST 1: SCENE 05 SLUGLINE & TOTAL SHOTS ---")
    
    # Locate Scene 05 start and end
    sc05_start = tap01.find("#### CẢNH 05:")
    sc06_start = tap01.find("#### CẢNH 06:")
    if sc05_start == -1 or sc06_start == -1 or sc06_start <= sc05_start:
        test_results["failures"].append("Could not delineate Scene 05 boundaries in TAP_01")
        print("FAIL: Could not delineate Scene 05 boundaries")
        return False
    
    sc05_text = tap01[sc05_start:sc06_start]
    
    # Extract all shots in Scene 05
    shot_pattern = re.compile(r'######\s+Shot\s+(\d+):\s+`?(ep01_scene05_shot(\d+))`?\s+\((\d+)s\)', re.IGNORECASE)
    sc05_shots = list(shot_pattern.finditer(sc05_text))
    
    print(f"Total shot headings found in Scene 05: {len(sc05_shots)}")
    assert len(sc05_shots) == 27, f"Scene 05 must have exactly 27 shots, found {len(sc05_shots)}"
    
    # Verify strict contiguous numbering 01 to 27
    sc05_shot_ids = [m.group(2) for m in sc05_shots]
    expected_sc05_ids = [f"ep01_scene05_shot{i:02d}" for i in range(1, 28)]
    assert sc05_shot_ids == expected_sc05_ids, f"Scene 05 shot IDs must strictly match 01-27 contiguous. Got: {sc05_shot_ids}"
    
    # Verify duration of each shot is 10s
    for m in sc05_shots:
        dur = int(m.group(4))
        assert dur == 10, f"Shot {m.group(2)} has duration {dur}s, expected 10s"
    
    test_results["tests"]["sc05_total_shots"] = {
        "status": "PASS",
        "count": len(sc05_shots),
        "expected": 27,
        "contiguous": True,
        "total_duration_sec": 270
    }
    print("PASS: Scene 05 contains exactly 27 contiguous shots (270s = 4m30s).")

    # =========================================================================
    # 2. SCENE 05 3-BEAT STRUCTURE INTEGRITY (C05-A, C05-B, C05-C)
    # =========================================================================
    print("\n--- TEST 2: 3-BEAT SUB-SCENE PARTITION & SHOT ALLOCATION ---")
    
    pos_c05a = sc05_text.find("PHÂN ĐOẠN 05-A") if "PHÂN ĐOẠN 05-A" in sc05_text else sc05_text.find("##### CẢNH 05-A:")
    pos_c05b = sc05_text.find("PHÂN ĐOẠN 05-B") if "PHÂN ĐOẠN 05-B" in sc05_text else sc05_text.find("##### CẢNH 05-B:")
    pos_c05c = sc05_text.find("PHÂN ĐOẠN 05-C") if "PHÂN ĐOẠN 05-C" in sc05_text else sc05_text.find("##### CẢNH 05-C:")
    
    assert pos_c05a != -1, "CẢNH 05-A heading missing"
    assert pos_c05b != -1, "CẢNH 05-B heading missing"
    assert pos_c05c != -1, "CẢNH 05-C heading missing"
    assert pos_c05a < pos_c05b < pos_c05c, "3-Beat order violated (must be A -> B -> C)"
    
    text_c05a = sc05_text[pos_c05a:pos_c05b]
    text_c05b = sc05_text[pos_c05b:pos_c05c]
    text_c05c = sc05_text[pos_c05c:]
    
    shots_a = list(shot_pattern.finditer(text_c05a))
    shots_b = list(shot_pattern.finditer(text_c05b))
    shots_c = list(shot_pattern.finditer(text_c05c))
    
    print(f"C05-A shot count: {len(shots_a)} (expected 8: Shots 01-08)")
    print(f"C05-B shot count: {len(shots_b)} (expected 7: Shots 09-15)")
    print(f"C05-C shot count: {len(shots_c)} (expected 12: Shots 16-27)")
    
    assert len(shots_a) == 8, f"C05-A expected 8 shots, got {len(shots_a)}"
    assert len(shots_b) == 7, f"C05-B expected 7 shots, got {len(shots_b)}"
    assert len(shots_c) == 12, f"C05-C expected 12 shots, got {len(shots_c)}"
    
    ids_a = [m.group(2) for m in shots_a]
    ids_b = [m.group(2) for m in shots_b]
    ids_c = [m.group(2) for m in shots_c]
    
    assert ids_a == [f"ep01_scene05_shot{i:02d}" for i in range(1, 9)], f"C05-A IDs mismatch: {ids_a}"
    assert ids_b == [f"ep01_scene05_shot{i:02d}" for i in range(9, 16)], f"C05-B IDs mismatch: {ids_b}"
    assert ids_c == [f"ep01_scene05_shot{i:02d}" for i in range(16, 28)], f"C05-C IDs mismatch: {ids_c}"
    
    test_results["tests"]["beat_partition"] = {
        "status": "PASS",
        "C05-A": {"count": len(shots_a), "range": "shot01-shot08", "duration": "80s (1m20s)"},
        "C05-B": {"count": len(shots_b), "range": "shot09-shot15", "duration": "70s (1m10s)"},
        "C05-C": {"count": len(shots_c), "range": "shot16-shot27", "duration": "120s (2m00s)"},
        "sum_verified": len(shots_a) + len(shots_b) + len(shots_c) == 27
    }
    print("PASS: 3-Beat partition strictly matches: A(8) + B(7) + C(12) = 27 shots.")

    # =========================================================================
    # 3. ADVERSARIAL SEMANTIC AUDIT FOR C05-A (BEAT 1: TIỂU KHÊ DẪN CẢNH)
    # =========================================================================
    print("\n--- TEST 3: SEMANTIC AUDIT OF C05-A (TIỂU KHÊ DẪN CẢNH) ---")
    
    # Required motifs in C05-A:
    # 1. Tiểu khê / stream stroll ("tiểu khê", "suối nhỏ", "nao nao dòng nước")
    # 2. Footbridge ("dịp cầu", "nhịp cầu gỗ", "cuối ghềnh")
    # 3. Willows / dusk stroll ("thơ thẩn dan tay ra về", "tà tà bóng ngả", "chiều tà")
    # 4. Shot 08 transition to barren landscape / ominous premonition
    
    c05a_lower = text_c05a.lower()
    assert "tiểu khê" in c05a_lower or "suối" in c05a_lower, "Missing stream / tiểu khê in C05-A"
    assert "cầu" in c05a_lower, "Missing bridge motif in C05-A"
    assert "tà tà bóng ngả về tây" in c05a_lower or "bóng ngả về tây" in c05a_lower, "Missing dusk poetry anchor in C05-A"
    assert "dịp cầu nho nhỏ" in c05a_lower or "nhịp cầu gỗ" in c05a_lower, "Missing footbridge poetry anchor in C05-A"
    
    # Check shot-by-shot 4 stems audio presence in C05-A
    for s_id, s_text in zip(ids_a, [text_c05a[m.start():] for m in shots_a]):
        assert "Stem 1" in s_text, f"Shot {s_id} missing Stem 1"
        assert "Stem 2" in s_text, f"Shot {s_id} missing Stem 2"
        assert "Stem 3" in s_text, f"Shot {s_id} missing Stem 3"
        assert "Stem 4" in s_text, f"Shot {s_id} missing Stem 4"
    
    test_results["tests"]["c05a_semantics"] = {
        "status": "PASS",
        "motifs_present": ["tiểu khê", "nhịp cầu gỗ", "bóng ngả về tây", "4-stem audio in all 8 shots"]
    }
    print("PASS: C05-A correctly implements Beat 1 stream stroll, bridge, dusk, and 4-stem audio.")

    # =========================================================================
    # 4. ADVERSARIAL SEMANTIC AUDIT FOR C05-B (BEAT 2: SÈ SÈ NẤM ĐẤT & THẮC MẮC)
    # =========================================================================
    print("\n--- TEST 4: SEMANTIC AUDIT OF C05-B (SÈ SÈ NẤM ĐẤT & THẮC MẮC) ---")
    
    # Required motifs in C05-B:
    # 1. Mound ("sè sè nấm đất")
    # 2. Half-yellow grass ("dàu dàu ngọn cỏ" / "nửa vàng nửa xanh")
    # 3. Barren banyan / desolate setting ("gốc bàng", "không một tấc bia đá", "hoang vu")
    # 4. Kiều inquiry to Vương Quan ("sao trong tiết thanh minh", "hương khói vắng tanh")
    # 5. Anti-Male-Tears compliance for Vương Quan (Shot 13)
    # 6. Off-screen / listening lip-sync guard for Vương Quan (Shot 12)
    
    c05b_lower = text_c05b.lower()
    assert "sè sè nấm đất" in c05b_lower, "Missing 'sè sè nấm đất' in C05-B"
    assert "nửa vàng nửa xanh" in c05b_lower or "dàu dàu" in c05b_lower, "Missing 'dàu dàu / nửa vàng nửa xanh' in C05-B"
    assert "thanh minh" in c05b_lower and "vắng tanh" in c05b_lower, "Missing inquiry 'tiết thanh minh ... vắng tanh' in C05-B"
    assert "gốc bàng" in c05b_lower, "Missing old banyan tree anchor in C05-B"
    assert "khóa khẩu hình" in c05b_lower, "Missing lip-sync guard in C05-B (Shot 12 Vương Quan listening)"
    
    # Check Shot 13 specifically for Anti-Male-Tears
    shot13_match = re.search(r'###### Shot 13:.*?(\n######|\Z)', text_c05b, re.DOTALL)
    assert shot13_match is not None, "Could not extract Shot 13"
    shot13_text = shot13_match.group(0)
    assert "r4" in shot13_text.lower() or "anti-male-tears" in shot13_text.lower(), "Shot 13 missing R4 Anti-Male-Tears citation"
    assert "không rơi lệ" in shot13_text.lower() or "tuyệt đối không có biểu hiện yếu mềm" in shot13_text.lower() or "đanh lại" in shot13_text.lower()
    
    test_results["tests"]["c05b_semantics"] = {
        "status": "PASS",
        "motifs_present": ["sè sè nấm đất", "dàu dàu nửa vàng nửa xanh", "inquiry to Vương Quan", "lip-sync guard Shot 12", "Anti-male-tears Shot 13"]
    }
    print("PASS: C05-B correctly implements Beat 2 grave discovery, inquiry, and stoic male reaction.")

    # =========================================================================
    # 5. ADVERSARIAL SEMANTIC AUDIT FOR C05-C (BEAT 3: TÂM LÝ, KHÓC, KHẮC THƠ, DẤU HÀI)
    # =========================================================================
    print("\n--- TEST 5: SEMANTIC AUDIT OF C05-C (BIẾN CHUYỂN TÂM LÝ & DẤU HÀI HIỂN LINH) ---")
    
    c05c_lower = text_c05c.lower()
    
    # Check specific beats in C05-C:
    # 1. Vương Quan storytelling (Shot 16): ca nhi, gãy cành thiên hương
    assert "ca nhi" in c05c_lower and "thiên hương" in c05c_lower, "Missing ca nhi / thiên hương in Shot 16"
    
    # 2. Kiều weeping & transformation (Shots 17-20)
    assert "bàng hoàng" in c05c_lower, "Missing phase 1 bàng hoàng in C05-C"
    assert "ngấn lệ" in c05c_lower or "châu sa" in c05c_lower, "Missing phase 2 ngấn lệ / châu sa in C05-C"
    assert "quỳ sụp" in c05c_lower, "Missing phase 3 quỳ sụp in C05-C"
    assert "đau đớn thay phận đàn bà" in c05c_lower, "Missing phase 4 lamentation 'Đau đớn thay phận đàn bà' in C05-C"
    assert "bạc mệnh" in c05c_lower, "Missing 'bạc mệnh' in C05-C"
    
    # 3. Incense offering (Shot 21)
    assert "hương trầm" in c05c_lower or "thắp hương" in c05c_lower or "nén hương" in c05c_lower, "Missing incense offering in Shot 21"
    assert "chân nấm" in c05c_lower or "nấm mồ" in c05c_lower, "Missing incense placement on mound"
    
    # 4. Hairpin poem carving on tree bark (Shot 22)
    assert "trâm bạc" in c05c_lower, "Missing silver hairpin in Shot 22"
    assert "vạch da cây" in c05c_lower or "khắc" in c05c_lower, "Missing poem carving on bark in Shot 22"
    assert "nhựa cây" in c05c_lower, "Missing oozing tree sap metaphor in Shot 22"
    
    # 5. Sibling contrast & philosophical dispute (Shots 23-24)
    assert "dư nước mắt" in c05c_lower, "Missing Thúy Vân's line 'khéo dư nước mắt khóc người đời xưa'"
    assert "vận vào" in c05c_lower or "nói gở" in c05c_lower, "Missing Vương Quan's warning in Shot 24"
    
    # 6. Kiều spiritual conviction (Shot 25)
    assert "thác là thể phách" in c05c_lower and "tinh anh" in c05c_lower, "Missing 'Thác là thể phách, còn là tinh anh'"
    
    # 7. Supernatural whirlwind & footprints on moss (Shot 26)
    assert "gió xoáy" in c05c_lower or "gió cuốn" in c05c_lower, "Missing whirlwind in Shot 26"
    assert "dấu hài" in c05c_lower or "dấu giày" in c05c_lower, "Missing ghost footprints in Shot 26"
    assert "in rêu" in c05c_lower or "thảm rêu" in c05c_lower, "Missing footprint in moss in Shot 26"
    
    # 8. Epilogue & second poem (Shot 27)
    assert "hữu tình ta lại gặp ta" in c05c_lower or "hiển linh" in c05c_lower, "Missing spiritual recognition in Shot 27"
    assert "vạch" in c05c_lower and "cổ thi" in c05c_lower, "Missing second poem on bark in Shot 27"
    
    test_results["tests"]["c05c_semantics"] = {
        "status": "PASS",
        "sub_beats_verified": [
            "Shot 16: Vương Quan narrates Đạm Tiên ca nhi",
            "Shots 17-20: Kiều 5-stage transformation (shock -> tears -> collapse -> lamentation 'Đau đớn thay phận đàn bà')",
            "Shot 21: Incense offering with 3 incense sticks",
            "Shot 22: Silver hairpin poem carved on tree bark with oozing sap",
            "Shots 23-24: Sibling contrast & philosophical dispute (Vân 'dư nước mắt', Quan 'vận vào')",
            "Shot 25: Spiritual conviction 'thác là thể phách còn là tinh anh'",
            "Shot 26: Supernatural whirlwind & ghost footprints on moss ('dấu hài in rêu rành rành')",
            "Shot 27: Epilogue, gratitude & second poem on tree"
        ]
    }
    print("PASS: C05-C correctly implements all 8 psychological, ritual, philosophical, and supernatural beats.")

    # =========================================================================
    # 6. SCENE 08 SHOT 08 LIP-SYNC GUARD EMPIRICAL STRESS TEST
    # =========================================================================
    print("\n--- TEST 6: SCENE 08 SHOT 08 LIP-SYNC GUARD AUDIT ---")
    
    sc08_start = tap01.find("#### CẢNH 08:")
    sc09_start = tap01.find("#### CẢNH 09:")
    assert sc08_start != -1 and sc09_start != -1, "Could not locate Scene 08 boundaries"
    
    sc08_text = tap01[sc08_start:sc09_start]
    
    # Find Shot 08 within Scene 08
    shot08_header = "##### Shot 08: `ep01_scene08_shot08` (10s)"
    assert shot08_header in sc08_text, f"Missing header: {shot08_header}"
    
    shot08_pos = sc08_text.find(shot08_header)
    shot08_text = sc08_text[shot08_pos:]
    
    print("Extracting Shot 08 details...")
    # Check 1: In-frame character is Kim Trọng
    assert "kim trọng" in shot08_text.lower(), "Kim Trọng must be the in-frame subject in Shot 08"
    
    # Check 2: Kiều is off-screen / distant voice
    assert "thúy kiều" in shot08_text.lower() and ("o.s." in shot08_text.lower() or "từ xa" in shot08_text.lower() or "bên kia tường" in shot08_text.lower()), \
        "Thúy Kiều must be off-screen / calling from distance across wall"
    
    # Check 3: Closed lips guard directive presence
    closed_lips_directive = "[KHÓA KHẨU HÌNH BẮT BUỘC]: Đôi môi nhân vật trong khung hình khép chặt tự nhiên khi có tiếng gọi từ xa"
    assert closed_lips_directive in shot08_text, f"Exact closed lips directive missing from Scene 08 Shot 08!\nExpected: {closed_lips_directive}"
    
    # Check 4: Verify Kim Trọng does NOT speak on-screen in Shot 08
    # If Kim Trọng speaks on-screen in Shot 08, it violates the off-screen listening guard!
    kt_dialogue_in_shot08 = re.findall(r'KIM TRỌNG.*?:.*?"(.*?)"', shot08_text)
    print(f"Kim Trọng on-screen dialogue instances in Shot 08: {len(kt_dialogue_in_shot08)}")
    assert len(kt_dialogue_in_shot08) == 0, f"Kim Trọng must NOT have dialogue in Shot 08 (he is only listening)! Found: {kt_dialogue_in_shot08}"
    
    # Check 5: Verify Kim Trọng's response properly occurs in Scene 09 Shot 01
    sc09_text = tap01[sc09_start:tap01.find("#### CẢNH 10:")]
    assert "ep01_scene09_shot01" in sc09_text, "Scene 09 Shot 01 missing"
    assert "KIM TRỌNG" in sc09_text and "Tiểu sinh Kim Trọng may mắn nhặt được" in sc09_text, \
        "Kim Trọng must speak on-screen in Scene 09 Shot 01"
    
    test_results["tests"]["sc08_shot08_lipsync"] = {
        "status": "PASS",
        "shot_id": "ep01_scene08_shot08",
        "in_frame_character": "Kim Trọng",
        "off_screen_speaker": "Thúy Kiều (O.S. bên kia tường hoa)",
        "guard_directive_verbatim": closed_lips_directive,
        "kim_trong_on_screen_dialogue_in_shot08": 0,
        "response_deferred_to_sc09_shot01": True
    }
    print("PASS: Scene 08 Shot 08 strictly enforces closed-lips guard for Kim Trọng.")

    # =========================================================================
    # 7. CROSS-SCENE LIP-SYNC AUDIT ACROSS ALL 188 SHOTS
    # =========================================================================
    print("\n--- TEST 7: EXHAUSTIVE LIP-SYNC GUARD AUDIT ACROSS TAP_01 ---")
    
    all_guards = re.findall(r'\[KHÓA KHẨU HÌNH.*?\]', tap01)
    print(f"Total Lip-Sync Guard directives across TAP_01: {len(all_guards)}")
    for g in all_guards:
        print(f"  Directive found: {g}")
    
    assert len(all_guards) >= 4, f"Expected at least 4 explicit lip-sync guards in screenplay, found {len(all_guards)}"
    
    test_results["tests"]["all_lipsync_guards"] = {
        "status": "PASS",
        "total_guards_found": len(all_guards),
        "guards": all_guards
    }

    # =========================================================================
    # 8. MALE TEARS SCAN IN SCENE 05 AND SCENE 08
    # =========================================================================
    print("\n--- TEST 8: MALE TEARS AUDIT IN SCENE 05 & SCENE 08 ---")
    
    male_names = ["Kim Trọng", "Vương Quan", "Vương Ông"]
    weep_words = ["khóc", "rơi lệ", "nước mắt", "châu sa", "nghẹn ngào", "nấc nghẹn"]
    
    for sc_name, sc_data in [("Scene 05", sc05_text), ("Scene 08", sc08_text)]:
        for line in sc_data.split('\n'):
            for mn in male_names:
                if mn in line:
                    for ww in weep_words:
                        if ww in line.lower():
                            # Check if negated
                            if "không" not in line.lower() and "tuyệt đối" not in line.lower() and "0%" not in line.lower():
                                test_results["failures"].append(f"Potential male crying in {sc_name}: '{line.strip()}'")
    
    assert len(test_results["failures"]) == 0, f"Found male tears violations: {test_results['failures']}"
    test_results["tests"]["male_tears_sc05_sc08"] = {"status": "PASS", "violations": 0}
    print("PASS: 0 male tears in Scene 05 or Scene 08.")

    # =========================================================================
    # 9. EXPORT RESULTS
    # =========================================================================
    output_path = os.path.join(base_dir, 'tests', 'adversarial_c05_c08_results.json')
    with open(output_path, 'w', encoding='utf-8') as f:
        json.dump(test_results, f, indent=2, ensure_ascii=False)
    
    print(f"\nAll tests passed successfully! Detailed results exported to {output_path}")
    return True

if __name__ == '__main__':
    ok = run_adversarial_c05_c08_tests()
    sys.exit(0 if ok else 1)
