import os
import re
import sys
import json
from collections import Counter

# Ensure UTF-8 output on Windows
sys.stdout.reconfigure(encoding='utf-8')

tap01_path = r"c:\Projects\KieuStory\FilmMaker\TAP_01_XUAN_SAC_THE_NGUYEN_VA_GIONG_BAO_DOAN_TRUONG.md"
idx_path = r"c:\Projects\KieuStory\FilmMaker\INDEX_VA_DANH_MUC_CANH_QUAY.md"
sb_path = r"c:\Projects\KieuStory\03_Storyboards\Ep01_Gia_Bien_Storyboard.md"
epguide_path = r"c:\Projects\KieuStory\01_Scripts_And_Episodes\EPISODE_GUIDE_30MIN.md"

with open(tap01_path, 'r', encoding='utf-8') as f:
    tap01 = f.read()

with open(idx_path, 'r', encoding='utf-8') as f:
    idx = f.read()

with open(sb_path, 'r', encoding='utf-8') as f:
    sb = f.read()

with open(epguide_path, 'r', encoding='utf-8') as f:
    epguide = f.read()

# Pattern matching shot headings (5 or 6 hashes)
shot_pattern = re.compile(r'#+\s+Shot\s+(\d+):\s+`?(ep01_scene\d+_shot\d+)`?\s+\(10s\)')
matches = list(shot_pattern.finditer(tap01))

shots = []
for i, m in enumerate(matches):
    start = m.start()
    end = matches[i+1].start() if i+1 < len(matches) else len(tap01)
    text = tap01[start:end]
    shot_num = int(m.group(1))
    shot_id = m.group(2)
    
    # Check sections
    has_cam = ('- **Cỡ cảnh & Máy quay**:' in text) or ('**Cỡ cảnh' in text)
    has_vis = ('- **Thị giác**:' in text) or ('**Thị giác' in text)
    has_aud = ('- **Âm thanh 4 Stems**:' in text) or ('**Âm thanh' in text)
    has_stem1 = '*Stem 1' in text
    has_stem2 = '*Stem 2' in text
    has_stem3 = '*Stem 3' in text
    has_stem4 = '*Stem 4' in text
    has_thoai = ('- **Thoại & Khẩu hình**:' in text)
    
    shots.append({
        'shot_id': shot_id,
        'shot_num': shot_num,
        'char_len': len(text.strip()),
        'has_cam': has_cam,
        'has_vis': has_vis,
        'has_aud': has_aud,
        'has_all_4_stems': has_stem1 and has_stem2 and has_stem3 and has_stem4,
        'has_thoai': has_thoai,
        'text': text
    })

# Compute statistics
char_lengths = [s['char_len'] for s in shots]
min_len = min(char_lengths)
max_len = max(char_lengths)
avg_len = sum(char_lengths) / len(char_lengths)

# Missing components
missing_cam = [s['shot_id'] for s in shots if not s['has_cam']]
missing_vis = [s['shot_id'] for s in shots if not s['has_vis']]
missing_aud = [s['shot_id'] for s in shots if not s['has_aud']]
missing_stems = [s['shot_id'] for s in shots if not s['has_all_4_stems']]

# Duplicate shot texts check (near duplicates)
visual_texts = []
for s in shots:
    # extract visual text
    vis_match = re.search(r'- \*\*Thị giác\*\*:(.*?)(?=- \*\*|\n\n|#####|$)', s['text'], re.DOTALL)
    if vis_match:
        visual_texts.append((s['shot_id'], vis_match.group(1).strip()))
    else:
        visual_texts.append((s['shot_id'], ''))

vis_text_counts = Counter(t[1] for t in visual_texts)
duplicated_visuals = [t for t, count in vis_text_counts.items() if count > 1 and len(t) > 20]

# Scene Breakdown
scene_counts = {}
for s in shots:
    sc = int(re.search(r'ep01_scene(\d+)_', s['shot_id']).group(1))
    scene_counts[sc] = scene_counts.get(sc, 0) + 1

act1 = sum(scene_counts[i] for i in range(1, 5))
act2 = sum(scene_counts[i] for i in range(5, 10))
act3 = sum(scene_counts[i] for i in range(10, 16))

# Check placeholders
placeholders = re.findall(r'\b(TODO|TBD|FIXME|PLACEHOLDER|XXX|đang cập nhật|tự code|như trên)\b', tap01, re.I)

# Check male crying
# Check occurrences of crying words associated with male characters
male_names = ["Kim Trọng", "Vương Quan", "Vương Ông", "Từ Hải"]
crying_words = ["khóc", "rơi lệ", "nước mắt", "rưng rưng", "giọt lệ", "nghẹn ngào rơi", "khóc than"]

male_violations = []
for line_no, line in enumerate(tap01.splitlines(), 1):
    for male in male_names:
        if male in line:
            for cw in crying_words:
                if cw in line:
                    line_lower = line.lower()
                    # if negated or explicitly about female / object
                    if any(neg in line_lower for neg in ["không", "triệt tiêu", "0%", "nén", "chối từ", "bảo vệ"]):
                        continue
                    if "của kiều" in line_lower or "thúy kiều" in line_lower or "vương bà" in line_lower:
                        # check if male is just mentioned e.g. "trao cho Kim Trọng chiếc quạt... nước mắt Kiều"
                        if re.search(r'(kiều|vương bà)[^.\n]*?(nước mắt|rơi lệ|giọt lệ|khóc)', line_lower) or \
                           re.search(r'(nước mắt|giọt lệ)[^.\n]*?(kiều|của kiều)', line_lower):
                            continue
                    male_violations.append((line_no, male, cw, line.strip()))

# Output Report
report = {
    "total_shots": len(shots),
    "unique_shot_ids": len(set(s['shot_id'] for s in shots)),
    "char_length_stats": {
        "min": min_len,
        "max": max_len,
        "avg": round(avg_len, 1)
    },
    "completeness": {
        "missing_camera": missing_cam,
        "missing_visuals": missing_vis,
        "missing_audio": missing_aud,
        "missing_4_stems": missing_stems
    },
    "duplicated_visual_descriptions": len(duplicated_visuals),
    "placeholders_found": placeholders,
    "scene_shot_counts": scene_counts,
    "act_breakdown": {
        "Act 1 (Scenes 01-04)": act1,
        "Act 2 (Scenes 05-09)": act2,
        "Act 3 (Scenes 10-15)": act3,
        "Total": act1 + act2 + act3
    },
    "male_tears_violations": male_violations,
    "dam_tien_sequence": {
        "scene05_shots": scene_counts.get(5, 0),
        "C05-A_shots": len([s for s in shots if 'scene05_' in s['shot_id'] and s['shot_num'] <= 8]),
        "C05-B_shots": len([s for s in shots if 'scene05_' in s['shot_id'] and 9 <= s['shot_num'] <= 15]),
        "C05-C_shots": len([s for s in shots if 'scene05_' in s['shot_id'] and 16 <= s['shot_num'] <= 27]),
        "has_dau_hai_in_reu": "dấu hài" in tap01.lower() and "in rêu" in tap01.lower(),
        "has_tram_bac_khac_tho": "trâm bạc" in tap01.lower() and "vạch da cây" in tap01.lower()
    },
    "lip_sync_guard_shot08": "[KHÓA KHẨU HÌNH BẮT BUỘC R5]" in [s['text'] for s in shots if s['shot_id'] == 'ep01_scene08_shot08'][0],
    "storyboard_check": "tiếng Vương Ông kêu nghẹn ngào" not in sb,
    "epguide_check": "Tiếng khóc than xé lòng của gia đình Vương viên ngoại" not in epguide,
    "index_summary_verified": ("15 Phân cảnh | 188 Shots | Thời lượng: 29 phút 40 giây" in idx) and ("56 Phân cảnh điện ảnh lớn" in idx)
}

print(json.dumps(report, indent=2, ensure_ascii=False))
