import os
import re
import sys

sys.stdout.reconfigure(encoding='utf-8')

base_dir = os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))
tap01_path = os.path.join(base_dir, 'FilmMaker', 'TAP_01_XUAN_SAC_THE_NGUYEN_VA_GIONG_BAO_DOAN_TRUONG.md')

with open(tap01_path, 'r', encoding='utf-8') as f:
    tap01 = f.read()

# Pattern for dialogue: **CHARACTER_NAME (parenthetical)**: "Dialogue"
dialogue_pattern = re.compile(r'\*\*([A-ZÀ-Ỹ\s]+)\s*\((.*?)\)\*\*:\s*(.*)')
dialogues = dialogue_pattern.findall(tap01)

print(f"Total dialogues with parentheticals found: {len(dialogues)}")

male_names = ["KIM TRỌNG", "VƯƠNG QUAN", "VƯƠNG ÔNG", "CỤ ÔNG", "VIÊN QUAN SAI NHA", "SAI NHA", "KẺ LẠ MẶT"]

weeping_kws = ["khóc", "lệ", "nước mắt", "nghẹn ngào", "nấc", "sụt sùi", "châu sa"]

male_dialogue_count = 0
alerts = []
for char, paren, text in dialogues:
    char_clean = char.strip()
    is_male = any(m in char_clean for m in male_names)
    paren_has_weep = any(kw in paren.lower() for kw in weeping_kws)
    text_has_weep = any(kw in text.lower() for kw in weeping_kws)
    
    if is_male:
        male_dialogue_count += 1
        print(f"\n[MALE DIALOGUE] Character: {char_clean}")
        print(f"  Parenthetical: {paren}")
        print(f"  Text: {text[:100]}...")
        if paren_has_weep:
            alerts.append(f"Parenthetical weep alert in {char_clean}: {paren}")
            print(f"  *** ALERT: Parenthetical has weep keyword! ***")
        if text_has_weep:
            alerts.append(f"Dialogue text weep alert in {char_clean}: {text}")
            print(f"  *** ALERT: Dialogue text has weep keyword! ***")

print(f"\nTotal male dialogues inspected: {male_dialogue_count}")
print(f"Total alerts: {len(alerts)}")
if alerts:
    for a in alerts:
        print(f"ALERT: {a}")
else:
    print("ALL MALE DIALOGUES AND PARENTHETICALS ARE 100% CLEAN OF WEEPING/TEARS KEYWORDS.")
