import os
import re
import sys
import json

sys.stdout.reconfigure(encoding='utf-8')

base_dir = os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))
tap01_path = os.path.join(base_dir, 'FilmMaker', 'TAP_01_XUAN_SAC_THE_NGUYEN_VA_GIONG_BAO_DOAN_TRUONG.md')

with open(tap01_path, 'r', encoding='utf-8') as f:
    tap01 = f.read()

# Let's find every line in tap01 with weeping keywords:
weeping_keywords = [
    "khóc", "rơi lệ", "giọt lệ", "ngấn lệ", "đầm đìa", "châu sa", 
    "rơi nước mắt", "nước mắt", "nấc nghẹn", "sụt sùi", "thút thít", 
    "ứa lệ", "hoen lệ", "hàng lệ", "dòng lệ", "lệ rơi", "tuôn lệ", "lệ tràn", "đẫm lệ"
]

male_characters = [
    "Kim Trọng", "Kim sinh", "chàng Kim", "Kim", 
    "Vương Quan", "Quan", 
    "Vương Ông", "Vương viên ngoại", "Viên ngoại", "cụ ông", "người cha", "cha Kiều"
]

lines = tap01.split('\n')
print(f"Total lines in TAP_01: {len(lines)}")

print("\n--- DETAILED INSPECTION OF ALL LINES WITH WEEPING KEYWORDS ---")
for idx, line in enumerate(lines, 1):
    matched_kws = [kw for kw in weeping_keywords if kw in line.lower()]
    if matched_kws:
        # Check if any male character appears in this line or previous 2 lines or next 2 lines
        context_lines = lines[max(0, idx-3):min(len(lines), idx+2)]
        context_str = " | ".join(l.strip() for l in context_lines if l.strip())
        
        has_male = False
        matched_males = []
        for mc in male_characters:
            if mc in ["Quan", "Kim", "Ông"]:
                pat = r'(?<![a-zA-Z0-9_\u00C0-\u024F\u1EA0-\u1EF9])' + re.escape(mc) + r'(?![a-zA-Z0-9_\u00C0-\u024F\u1EA0-\u1EF9])'
            else:
                pat = re.escape(mc)
            if re.search(pat, line, re.IGNORECASE):
                # Filter out false positives
                if mc == "Quan" and re.search(r'(viên quan|quan nha|quan lại|sai nha|tổng quan|chức quan)', line, re.IGNORECASE) and not re.search(r'Vương Quan', line, re.IGNORECASE):
                    continue
                has_male = True
                matched_males.append(mc)

        print(f"\nLine {idx} [KWs: {matched_kws}] [Males: {matched_males if has_male else 'None'}]:")
        print(f"  TEXT: {line.strip()}")
