import re
import sys
import json
from pathlib import Path

sys.stdout.reconfigure(encoding='utf-8')

screenplay_path = Path("FilmMaker/TAP_01_XUAN_SAC_THE_NGUYEN_VA_GIONG_BAO_DOAN_TRUONG.md")
content = screenplay_path.read_text(encoding="utf-8")

# Extract scene headers
scene_matches = list(re.finditer(r'####\s+CẢNH\s+(\d+):\s*([^\n]+)', content))
print(f"Total scenes found: {len(scene_matches)}")

# Split into shots
shot_pattern = re.compile(
    r'#####\s+Shot\s+(\d+):\s*[`\x60](ep01_scene\d+_shot\d+)[`\x60]\s*\(([^)]+)\)\n'
    r'-\s*\*\*Cỡ cảnh & Máy quay\*\*:\s*([^\n]+)\n'
    r'-\s*\*\*Thị giác\*\*:\s*(.*?)\n'
    r'-\s*\*\*Âm thanh 4 Stems\*\*:\s*\n'
    r'(.*?)(?=(?:#####\s+Shot|####\s+CẢNH|\Z))',
    re.DOTALL
)

shots = list(shot_pattern.finditer(content))
print(f"Total shots parsed via regex: {len(shots)}")

if len(shots) != 188:
    print(f"Warning: parsed {len(shots)} instead of 188. Let's inspect differences.")
    # find all shot IDs
    all_shot_ids = re.findall(r'[`\x60](ep01_scene\d+_shot\d+)[`\x60]', content)
    unique_ids = list(dict.fromkeys(all_shot_ids))
    print(f"Unique shot IDs in file: {len(unique_ids)}")
    parsed_ids = [m.group(2) for m in shots]
    diff = set(unique_ids) - set(parsed_ids)
    print(f"Unparsed IDs: {diff}")
