"""
Test package initialization and shared test helpers for KieuStory E2E test suite.
"""

import os
import sys
import re
import json
import subprocess
import tempfile
from pathlib import Path
from typing import Dict, List, Optional, Tuple, Any

BASE_DIR = Path(__file__).resolve().parent.parent
FILMMAKER_DIR = BASE_DIR / "FilmMaker"
PROMPTS_DIR = BASE_DIR / "02_AI_Prompts"
ASSETS_DIR = BASE_DIR / "04_Assets"
BIBLE_DIR = BASE_DIR / "00_Project_Bible"
PIPELINE_DIR = BASE_DIR / "05_Production_Pipeline"
EXPORTS_DIR = BASE_DIR / "06_Exports"
WEB_REVIEW_DIR = BASE_DIR / "web_review"

# Ensure pipeline and web_review are on sys.path
for p in [str(PIPELINE_DIR), str(WEB_REVIEW_DIR), str(BASE_DIR)]:
    if p not in sys.path:
        sys.path.insert(0, p)

MALE_CHARACTERS = [
    "KIM TRỌNG", "VƯƠNG QUAN", "VƯƠNG ÔNG", "TỪ HẢI",
    "THÚC SINH", "THÚC ÔNG", "MÃ GIÁM SINH", "SỞ KHANH",
    "HỒ TÔN HIẾN", "BẠC BÀ", "BẠC HẠNH"
]

CRYING_KEYWORDS = [
    "khóc", "rơi lệ", "rơi nước mắt", "châu sa", "ngấn lệ",
    "nước mắt", "nước mắt tuôn", "khóc ngất", "khóc rống",
    "gào khóc", "nức nở", "thút thít", "rưng rưng giọt lệ"
]


def load_muse_prompts_data() -> Dict[str, Any]:
    prompts_path = PROMPTS_DIR / "muse_ai_video_prompts.json"
    if prompts_path.exists():
        with open(prompts_path, "r", encoding="utf-8") as f:
            return json.load(f)
    return {}


def load_banana_prompts_data() -> Dict[str, Any]:
    prompts_path = PROMPTS_DIR / "gemini_banana_prompts.json"
    if prompts_path.exists():
        with open(prompts_path, "r", encoding="utf-8") as f:
            return json.load(f)
    return {}


def read_ep01_screenplay() -> str:
    path = FILMMAKER_DIR / "TAP_01_XUAN_SAC_THE_NGUYEN_VA_GIONG_BAO_DOAN_TRUONG.md"
    if path.exists():
        return path.read_text(encoding="utf-8")
    return ""


def get_screenplay_scenes(text: str) -> List[Tuple[str, str]]:
    """Splits screenplay into (scene_title, scene_content)."""
    scenes = []
    # Match headers like #### CẢNH 01: ... or ### CẢNH 01: ...
    parts = re.split(r"(#{3,4}\s*CẢNH\s*\d+[^:\n]*:[^\n]*)", text)
    if len(parts) > 1:
        for i in range(1, len(parts), 2):
            header = parts[i].strip()
            content = parts[i + 1] if i + 1 < len(parts) else ""
            scenes.append((header, content))
    return scenes


def find_male_crying_violations(screenplay_text: str) -> List[Dict[str, Any]]:
    """
    Scans screenplay for male characters associated with crying action beats.
    Returns list of violations with character, line number, and text snippet.
    """
    violations = []
    lines = screenplay_text.splitlines()
    current_speaker = None

    for idx, line in enumerate(lines, 1):
        line_clean = line.strip()
        if not line_clean:
            continue

        # Check speaker line like **KIM TRỌNG** or KIM TRỌNG:
        speaker_match = re.match(r"^\*{0,2}([A-ZÀ-Ỹ\s]+)\*{0,2}:?", line_clean)
        if speaker_match:
            potential_speaker = speaker_match.group(1).strip()
            for mc in MALE_CHARACTERS:
                if mc in potential_speaker:
                    current_speaker = mc
                    break
            else:
                if any(fc in potential_speaker for fc in ["THÚY KIỀU", "THÚY VÂN", "VƯƠNG BÀ", "ĐẠM TIÊN", "HOẠN THƯ"]):
                    current_speaker = "FEMALE"
                elif "NGƯỜI DẪN CHUYỆN" in potential_speaker or "V.O." in potential_speaker:
                    current_speaker = "NARRATOR"

        # Check action beats / parentheticals for current male speaker
        if current_speaker in MALE_CHARACTERS:
            # Action line or parenthetical
            for kw in CRYING_KEYWORDS:
                if kw in line_clean.lower():
                    violations.append({
                        "line_num": idx,
                        "character": current_speaker,
                        "keyword": kw,
                        "text": line_clean
                    })
                    break

        # Also check direct inline patterns e.g. "KIM TRỌNG *(Nắm chặt tay Kiều, nghẹn ngào rơi nước mắt)*"
        for mc in MALE_CHARACTERS:
            if mc in line_clean:
                # Exclude if the tears are explicitly attributed to a female in the scene
                is_female_tears = any(f in line_clean.lower() for f in ["thiếu nữ", "nàng", "cô gái", "người thiếu nữ", "kiều"]) and any(c in line_clean.lower() for c in ["quỳ trong nước mắt", "mắt nàng", "nước mắt nàng"])
                if is_female_tears:
                    continue
                for kw in CRYING_KEYWORDS:
                    if kw in line_clean.lower():
                        if not any(v["line_num"] == idx for v in violations):
                            violations.append({
                                "line_num": idx,
                                "character": mc,
                                "keyword": kw,
                                "text": line_clean
                            })
                            break

    return violations


def create_synthetic_wav(path: str, duration_sec: float = 2.0, sample_rate: int = 48000, freq: float = 440.0):
    """Generates a synthetic sine wave WAV file for testing without external deps."""
    import struct
    import math

    num_samples = int(duration_sec * sample_rate)
    num_channels = 2
    bits_per_sample = 16
    byte_rate = sample_rate * num_channels * bits_per_sample // 8
    block_align = num_channels * bits_per_sample // 8
    data_size = num_samples * block_align
    chunk_size = 36 + data_size

    with open(path, "wb") as f:
        # RIFF header
        f.write(b"RIFF")
        f.write(struct.pack("<I", chunk_size))
        f.write(b"WAVE")
        # fmt subchunk
        f.write(b"fmt ")
        f.write(struct.pack("<I", 16))
        f.write(struct.pack("<H", 1)) # PCM
        f.write(struct.pack("<H", num_channels))
        f.write(struct.pack("<I", sample_rate))
        f.write(struct.pack("<I", byte_rate))
        f.write(struct.pack("<H", block_align))
        f.write(struct.pack("<H", bits_per_sample))
        # data subchunk
        f.write(b"data")
        f.write(struct.pack("<I", data_size))
        for i in range(num_samples):
            val = int(32767.0 * 0.5 * math.sin(2.0 * math.pi * freq * (i / sample_rate)))
            packed = struct.pack("<hh", val, val)
            f.write(packed)


def create_synthetic_video_mp4(path: str, duration_sec: float = 2.0, width: int = 1280, height: int = 720):
    """Creates a minimal valid MP4 video with an AAC audio track using FFmpeg."""
    from audio_continuity_engine import get_ffmpeg
    ffmpeg = get_ffmpeg()
    cmd = [
        ffmpeg, "-y",
        "-f", "lavfi", "-i", f"color=c=black:s={width}x{height}:d={duration_sec}",
        "-f", "lavfi", "-i", f"sine=frequency=440:sample_rate=48000:duration={duration_sec}",
        "-c:v", "libx264", "-t", str(duration_sec), "-pix_fmt", "yuv420p",
        "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
        path
    ]
    res = subprocess.run(cmd, capture_output=True, text=True)
    return res.returncode == 0
