"""
=============================================================================
KIEU STORY AI CINEMA - STEM 2 CONTINUOUS FESTIVAL AMBIENCE GENERATOR
=============================================================================
Tạo asset Stem 2: Âm cảnh đám đông hội Đạp Thanh / Tiết Thanh Minh 120s
- Chuẩn: 48,000 Hz, 16-bit PCM Stereo WAV, thời lượng 120.0s (seamless bed).
- Đích: 04_Assets/audio_sfx/ep01_scene04_festival_crowd_ambience_120s.wav
  và 04_Assets/audio/festival_crowd_ambience_bed.wav
=============================================================================
"""

import os
import sys
import math
import struct
import random
from pathlib import Path

# Khởi tạo UTF-8 cho Windows Console
if sys.platform == "win32":
    try:
        sys.stdout.reconfigure(encoding="utf-8")
        sys.stderr.reconfigure(encoding="utf-8")
    except Exception:
        pass


def generate_festival_ambience(
    output_path: str,
    duration_sec: float = 120.0,
    sample_rate: int = 48000
) -> bool:
    """
    Tổng hợp track âm cảnh Stem 2 không gian hội xuân Thanh Minh:
    1. Đám đông rầm rì (Pink noise + formant filter 300Hz - 2500Hz).
    2. Gió xuân thoảng qua rặng liễu (LFO 0.12Hz panning).
    3. Chuông đồng / khánh ngọc xa xa ngân nga (pentatonic bells 587Hz, 880Hz).
    4. Giai điệu sáo trúc cổ phong thoang thoảng theo gió.
    5. Xao động áo lụa và tiếng bước chân tản bộ thanh nhã.
    """
    out_file = Path(output_path)
    out_file.parent.mkdir(parents=True, exist_ok=True)

    total_samples = int(duration_sec * sample_rate)
    num_channels = 2
    bits_per_sample = 16
    byte_rate = sample_rate * num_channels * bits_per_sample // 8
    block_align = num_channels * bits_per_sample // 8
    data_size = total_samples * block_align
    chunk_size = 36 + data_size

    print(f"[*] Đang tổng hợp Stem 2 Festival Ambience ({duration_sec}s, {sample_rate}Hz Stereo)...")

    # Pink noise filter states
    b0_l = b1_l = b2_l = b3_l = b4_l = b5_l = b6_l = 0.0
    b0_r = b1_r = b2_r = b3_r = b4_r = b5_r = b6_r = 0.0

    # Lowpass states
    lp_l, lp_r = 0.0, 0.0

    # Bamboo flute melodic notes (pentatonic frequencies in Hz)
    flute_notes = [293.66, 369.99, 440.00, 493.88, 587.33, 659.25, 880.00]

    random.seed(188)  # Deterministic synthesis for repeatable quality

    with open(out_file, "wb") as f:
        # RIFF Header
        f.write(b"RIFF")
        f.write(struct.pack("<I", chunk_size))
        f.write(b"WAVE")
        # fmt chunk
        f.write(b"fmt ")
        f.write(struct.pack("<I", 16))
        f.write(struct.pack("<H", 1))  # PCM
        f.write(struct.pack("<H", num_channels))
        f.write(struct.pack("<I", sample_rate))
        f.write(struct.pack("<I", byte_rate))
        f.write(struct.pack("<H", block_align))
        f.write(struct.pack("<H", bits_per_sample))
        # data chunk
        f.write(b"data")
        f.write(struct.pack("<I", data_size))

        buffer = bytearray()
        batch_size = 4800  # 0.1s batches

        for n in range(total_samples):
            t = n / sample_rate

            # White noise samples
            w_l = random.uniform(-1.0, 1.0)
            w_r = random.uniform(-1.0, 1.0)

            # Paul Kellet's Pink Noise Filter (Left)
            b0_l = 0.99886 * b0_l + w_l * 0.0555179
            b1_l = 0.99332 * b1_l + w_l * 0.0750759
            b2_l = 0.96900 * b2_l + w_l * 0.1538520
            b3_l = 0.86650 * b3_l + w_l * 0.3104856
            b4_l = 0.55000 * b4_l + w_l * 0.5329522
            b5_l = -0.7616 * b5_l - w_l * 0.0168980
            pink_l = (b0_l + b1_l + b2_l + b3_l + b4_l + b5_l + b6_l + w_l * 0.5362) * 0.11
            b6_l = w_l * 0.115926

            # Paul Kellet's Pink Noise Filter (Right)
            b0_r = 0.99886 * b0_r + w_r * 0.0555179
            b1_r = 0.99332 * b1_r + w_r * 0.0750759
            b2_r = 0.96900 * b2_r + w_r * 0.1538520
            b3_r = 0.86650 * b3_r + w_r * 0.3104856
            b4_r = 0.55000 * b4_r + w_r * 0.5329522
            b5_r = -0.7616 * b5_r - w_r * 0.0168980
            pink_r = (b0_r + b1_r + b2_r + b3_r + b4_r + b5_r + b6_r + w_r * 0.5362) * 0.11
            b6_r = w_r * 0.115926

            # Crowd lowpass filter (warm vocal murmur)
            lp_l += 0.08 * (pink_l - lp_l)
            lp_r += 0.08 * (pink_r - lp_r)

            # Spring wind gentle swell (0.12 Hz LFO)
            wind_l = 0.75 + 0.25 * math.sin(2.0 * math.pi * 0.12 * t)
            wind_r = 0.75 + 0.25 * math.cos(2.0 * math.pi * 0.12 * t)

            # Crowd chatter murmur dynamics (0.28 Hz & 0.45 Hz)
            crowd_mod = 0.82 + 0.18 * math.sin(2.0 * math.pi * 0.28 * t + 0.4)

            # Festival chimes / temple bells (period = 8.0s)
            bell_cycle = t % 8.0
            bell_env = math.exp(-bell_cycle * 2.2) if bell_cycle < 3.5 else 0.0
            chime_tone = bell_env * (
                0.018 * math.sin(2.0 * math.pi * 587.33 * t) +
                0.012 * math.sin(2.0 * math.pi * 880.00 * t) +
                0.006 * math.sin(2.0 * math.pi * 1174.66 * t)
            )

            # Distant bamboo flute melody (gentle, breathing every 6.0s)
            flute_cycle = t % 6.0
            note_idx = int(t / 3.0) % len(flute_notes)
            flute_freq = flute_notes[note_idx]
            flute_env = (math.sin(math.pi * (flute_cycle / 6.0)) ** 2) * 0.014 if flute_cycle < 5.0 else 0.0
            # Sáo trúc có vibrato 5Hz
            vibrato = 1.0 + 0.015 * math.sin(2.0 * math.pi * 5.0 * t)
            flute_tone = flute_env * (
                math.sin(2.0 * math.pi * flute_freq * vibrato * t) +
                0.3 * math.sin(4.0 * math.pi * flute_freq * vibrato * t)
            )

            # Combine left & right channels
            sig_l = (lp_l * wind_l * crowd_mod + chime_tone * 0.8 + flute_tone * 0.9) * 0.38
            sig_r = (lp_r * wind_r * crowd_mod + chime_tone * 0.6 + flute_tone * 0.7) * 0.38

            # Quantize to 16-bit PCM
            val_l = max(-32768, min(32767, int(sig_l * 32767.0)))
            val_r = max(-32768, min(32767, int(sig_r * 32767.0)))

            buffer.extend(struct.pack("<hh", val_l, val_r))

            if len(buffer) >= batch_size * 4:
                f.write(buffer)
                buffer.clear()

        if buffer:
            f.write(buffer)

    print(f"[✓] Đã tạo thành công Stem 2 Festival Ambience: {out_file} ({out_file.stat().st_size / (1024*1024):.2f} MB)")
    return True


if __name__ == "__main__":
    target = r"04_Assets/audio_sfx/ep01_scene04_festival_crowd_ambience_120s.wav"
    if len(sys.argv) > 1:
        target = sys.argv[1]
    dur = float(sys.argv[2]) if len(sys.argv) > 2 else 120.0
    generate_festival_ambience(target, duration_sec=dur)
