"""
=============================================================================
KIEU STORY AI CINEMA - AUDIO CONTINUITY & SOUNDSCAPE ENGINE (UPGRADED M3)
=============================================================================
Công cụ điều phối và chuẩn hóa âm thanh đồng nhất giữa các shot video AI (Muse.ai)
- 4-Stem Audio Architecture (Stem 1: BGM, Stem 2: Ambience, Stem 3: Foley, Stem 4: Dialogue).
- Equal-Power Audio Crossfade (qsin/cbrt) triệt tiêu tiếng click/pop ngắt cụt.
- Boundary Micro-Fade Mode A (30ms curve=qsin zero duration loss) & Mode B (acrossfade with tail padding).
- Stream Pre-conditioning: aresample=48000,aformat=channel_layouts=stereo và fallback aevalsrc=0 cho video câm.
- Active RMS Gain Staging: bù trừ âm lượng khi chênh lệch vượt 3.0 dB về mốc -24.0 dBFS kèm headroom guard.
- Xử lý gối đầu âm thanh qua đoạn nối hình ảnh (Transition Audio Bridge).
- Kỹ thuật J-Cut / L-Cut không gian (Acoustic Distance Filter 3500Hz) cho tiếng đàn Kiều.
- Dynamic Sidechain Ducking (hạ BGM & Ambience -14dB khi có thoại / ngâm thơ).
- Chuẩn hóa âm lượng EBU R128 Two-Pass Linear (-14 LUFS, TP -1.0 dBTP, LRA 9-11 LU) đạt chuẩn phát sóng YouTube Green Dollar.
=============================================================================
"""

import os
import sys
import re
import json
import argparse
import subprocess
import tempfile
import shutil
from pathlib import Path
from typing import List, Dict, Optional, Tuple, Any, Union

# Khởi tạo UTF-8 cho Windows Console
if sys.platform == "win32":
    try:
        sys.stdout.reconfigure(encoding="utf-8")
        sys.stderr.reconfigure(encoding="utf-8")
    except Exception:
        pass

# Tìm đường dẫn FFmpeg trong hệ thống
DEFAULT_FFMPEG_PATHS = [
    r"C:\Users\Admin\AppData\Local\Microsoft\WinGet\Packages\Gyan.FFmpeg_Microsoft.Winget.Source_8wekyb3d8bbwe\ffmpeg-9.0.2-full_build\bin\ffmpeg.exe",
    r"C:\Users\Admin\AppData\Local\Microsoft\WindowsApps\ffmpeg.exe",
    r"ffmpeg"
]
DEFAULT_FFPROBE_PATHS = [
    r"C:\Users\Admin\AppData\Local\Microsoft\WinGet\Packages\Gyan.FFmpeg_Microsoft.Winget.Source_8wekyb3d8bbwe\ffmpeg-9.0.2-full_build\bin\ffprobe.exe",
    r"ffprobe"
]

def get_ffmpeg() -> str:
    for p in DEFAULT_FFMPEG_PATHS:
        if os.path.exists(p) or p == "ffmpeg":
            try:
                res = subprocess.run([p, "-version"], capture_output=True, text=True)
                if res.returncode == 0:
                    return p
            except Exception:
                continue
    raise FileNotFoundError("Không tìm thấy ffmpeg.exe trong hệ thống.")

def get_ffprobe() -> str:
    for p in DEFAULT_FFPROBE_PATHS:
        if os.path.exists(p) or p == "ffprobe":
            try:
                res = subprocess.run([p, "-version"], capture_output=True, text=True)
                if res.returncode == 0:
                    return p
            except Exception:
                continue
    raise FileNotFoundError("Không tìm thấy ffprobe.exe trong hệ thống.")


class AudioContinuityEngine:
    # 4-Stem Definitions theo Project Bible CINEMATIC_AUDIO_PIPELINE.md
    STEM_1_BGM = "Stem 1: BGM & Score"
    STEM_2_AMBIENCE = "Stem 2: Ambience & Soundscape"
    STEM_3_FOLEY = "Stem 3: Foley & Spot SFX"
    STEM_4_DIALOGUE = "Stem 4: Dialogue & Voice-Over"

    def __init__(self):
        self.ffmpeg = get_ffmpeg()
        self.ffprobe = get_ffprobe()

    def inspect_shot_audio(self, video_path: str) -> Dict:
        """Phân tích chi tiết luồng âm thanh và mức âm lượng của một shot video."""
        if not os.path.exists(video_path):
            return {
                "file": video_path,
                "has_audio": False,
                "max_volume_db": -99.0,
                "mean_volume_db": -99.0,
                "error": "File not found"
            }

        # Probe streams
        probe_cmd = [
            self.ffprobe, "-v", "error",
            "-show_entries", "stream=codec_type,codec_name,sample_rate,channels,duration,bit_rate:format=duration",
            "-of", "json", video_path
        ]
        res = subprocess.run(probe_cmd, capture_output=True, text=True)
        probe_data = json.loads(res.stdout) if res.returncode == 0 else {}
        streams = probe_data.get("streams", [])
        fmt = probe_data.get("format", {})

        video_dur = float(fmt.get("duration", 0.0))
        audio_streams = [s for s in streams if s.get("codec_type") == "audio"]

        if not audio_streams:
            return {
                "file": os.path.basename(video_path),
                "has_audio": False,
                "duration": video_dur,
                "max_volume_db": -99.0,
                "mean_volume_db": -99.0
            }

        astream = audio_streams[0]
        # Volumedetect
        vol_cmd = [self.ffmpeg, "-i", video_path, "-af", "volumedetect", "-f", "null", "-"]
        vol_res = subprocess.run(vol_cmd, capture_output=True, text=True)
        stderr = vol_res.stderr

        max_vol, mean_vol = -99.0, -99.0
        for line in stderr.splitlines():
            if "max_volume:" in line:
                try:
                    max_vol = float(line.split("max_volume:")[1].split("dB")[0].strip())
                except Exception:
                    pass
            elif "mean_volume:" in line:
                try:
                    mean_vol = float(line.split("mean_volume:")[1].split("dB")[0].strip())
                except Exception:
                    pass

        adur = float(astream.get("duration", video_dur))
        return {
            "file": os.path.basename(video_path),
            "has_audio": True,
            "codec": astream.get("codec_name"),
            "sample_rate": int(astream.get("sample_rate", 48000)),
            "channels": int(astream.get("channels", 2)),
            "duration": adur if adur > 0 else video_dur,
            "max_volume_db": max_vol,
            "mean_volume_db": mean_vol
        }

    def inspect_video_dimensions(self, video_path: str) -> Tuple[int, int]:
        """Trích xuất độ phân giải (width, height) của video qua ffprobe."""
        if not os.path.exists(video_path):
            return 0, 0
        try:
            cmd = [
                self.ffprobe, "-v", "error",
                "-select_streams", "v:0",
                "-show_entries", "stream=width,height",
                "-of", "json", video_path
            ]
            res = subprocess.run(cmd, capture_output=True, text=True, timeout=5)
            if res.returncode == 0 and res.stdout:
                data = json.loads(res.stdout)
                streams = data.get("streams", [])
                if streams:
                    return int(streams[0].get("width", 0)), int(streams[0].get("height", 0))
        except Exception:
            pass
        return 0, 0

    def inspect_sequence(self, shot_paths: List[str]):
        """Kiểm tra tính liên tục âm thanh cho toàn bộ chuỗi shot."""
        print("\n" + "=" * 75)
        print("📊 KIỂM TRA ĐỘ ĐỒNG NHẤT ÂM THANH CHUỖI SHOT (AUDIO CONTINUITY AUDIT)")
        print("=" * 75)

        reports = []
        for p in shot_paths:
            rep = self.inspect_shot_audio(p)
            reports.append(rep)

        for i, r in enumerate(reports):
            status = "🔊 Có âm thanh" if r["has_audio"] else "🔇 CÂM (No Audio Track)"
            print(f"[{i+1}] {r['file']:<35} | {status:<15} | Max: {r['max_volume_db']:>5.1f} dB | Mean: {r['mean_volume_db']:>5.1f} dB")
            if i > 0:
                prev = reports[i-1]
                if r["has_audio"] and prev["has_audio"]:
                    delta_mean = abs(r["mean_volume_db"] - prev["mean_volume_db"])
                    if delta_mean > 3.0:
                        print(f"    ⚠️ CẢNH BÁO LỆCH ÂM LƯỢNG: Chênh {delta_mean:.1f} dB giữa Shot {i} và Shot {i+1} (Cần Audio Normalize)")
                    else:
                        print(f"    ✓ Độ chênh âm lượng an toàn: {delta_mean:.1f} dB")
                elif not r["has_audio"] and prev["has_audio"]:
                    print(f"    ⚠️ CẢNH BÁO MẤT TIẾNG: Shot {i+1} không có âm thanh (Cần Bridge Audio Crossfade)")

        print("=" * 75 + "\n")
        return reports

    def stitch_with_audio_crossfade(
        self,
        video_paths: List[str],
        output_path: str,
        crossfade_dur: float = 1.0,
        normalize_lufs: bool = True,
        mode: str = "acrossfade",
        active_gain_staging: bool = True
    ) -> bool:
        """
        Ghép nối chuỗi video với chuyển tiếp âm thanh mượt mà (Equal-Power Crossfade)
        và chuẩn hóa âm lượng EBU R128 (-14 LUFS) để tránh giật cụt.
        - Mode A ("boundary_smoothing" / "micro_crossfade"): Micro-fade 30ms (qsin) bảo toàn thời lượng 100%.
        - Mode B ("acrossfade" - Mặc định): Chuỗi acrossfade (c1=qsin, c2=qsin) kèm tail padding apad.
        - Stream Pre-conditioning: aresample=48000,aformat=channel_layouts=stereo và fallback aevalsrc=0 cho video câm.
        - Active RMS Gain Staging: Điều chỉnh các shot chênh lệch âm lượng vượt 3.0 dB về chuẩn -24.0 dBFS.
        """
        if not video_paths:
            print("[!] Danh sách file rỗng.")
            return False

        # HARD GATE: Khóa cứng tỷ lệ 16:9 Cinema Wide, cấm tuyệt đối video dọc (9:16)
        for p in video_paths:
            w, h = self.inspect_video_dimensions(p)
            if w > 0 and h > 0 and (w / h < 1.33):
                raise ValueError(
                    f"[CRITICAL ASPECT RATIO REJECTION] Video {os.path.basename(p)} có tỷ lệ dọc {w}x{h} (ratio={w/h:.3f}). "
                    f"Bản Master điện ảnh 16:9 NGHIÊM CẤM video 9:16. Ghép nối bị hủy! Cần render lại shot 16:9."
                )

        print(f"\n[*] Đang ghép {len(video_paths)} shot với Audio Crossfade ({crossfade_dur}s, mode={mode})...")
        os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True)

        inputs = []
        for p in video_paths:
            inputs.extend(["-i", p])

        filter_parts = []
        # Chuẩn hóa video stream về cùng chuẩn 1280x720, 24fps, SAR 1:1 (pillarbox/letterbox nếu cần)
        for i in range(len(video_paths)):
            filter_parts.append(
                f"[{i}:v]scale=1280:720:force_original_aspect_ratio=decrease,pad=1280:720:(ow-iw)/2:(oh-ih)/2,setsar=1,fps=24[v{i}_norm]"
            )
        # Nối video
        v_concat_inputs = "".join([f"[v{i}_norm]" for i in range(len(video_paths))])
        filter_parts.append(f"{v_concat_inputs}concat=n={len(video_paths)}:v=1:a=0[v_raw]")

        # 1. Stream Pre-conditioning & Active RMS Gain Staging
        reports = [self.inspect_shot_audio(p) for p in video_paths]

        for i, rep in enumerate(reports):
            if rep["has_audio"]:
                gain_filter = ""
                if active_gain_staging and rep["mean_volume_db"] > -90.0:
                    delta_target = -24.0 - rep["mean_volume_db"]
                    gain_adj = max(-12.0, min(6.0, delta_target))
                    # Headroom protection
                    if rep["max_volume_db"] + gain_adj > -1.0:
                        gain_adj = -1.0 - rep["max_volume_db"]
                    if abs(gain_adj) >= 0.2:
                        gain_filter = f"volume={gain_adj:.2f}dB,"

                # Resample 48000 Hz và chuẩn hóa stereo layout
                filter_parts.append(
                    f"[{i}:a]{gain_filter}aresample=48000,aformat=sample_rates=48000:channel_layouts=stereo[a{i}_norm]"
                )
            else:
                # Video câm: Tự động sinh audio silence fallback
                clip_dur = rep.get("duration", 10.0)
                if clip_dur <= 0:
                    clip_dur = 10.0
                filter_parts.append(f"aevalsrc=0:d={clip_dur}:s=48000:c=stereo[a{i}_norm]")

        # 2. Xử lý ghép âm thanh theo mode
        if len(video_paths) == 1:
            filter_parts.append("[a0_norm]anull[a_faded]")
        elif mode in ("boundary_smoothing", "micro_crossfade", "mode_a"):
            # Mode A: Sample-Accurate Boundary Micro-Crossfade (30ms curve=qsin)
            micro_dur = 0.030
            for i, rep in enumerate(reports):
                cdur = rep.get("duration", 10.0)
                if cdur <= 0:
                    cdur = 10.0
                out_start = max(0.0, cdur - micro_dur)
                filter_parts.append(
                    f"[a{i}_norm]afade=t=in:st=0:d={micro_dur}:curve=qsin,"
                    f"afade=t=out:st={out_start:.3f}:d={micro_dur}:curve=qsin[a{i}_faded]"
                )
            a_inputs = "".join([f"[a{i}_faded]" for i in range(len(video_paths))])
            filter_parts.append(f"{a_inputs}concat=n={len(video_paths)}:v=0:a=1[a_faded]")
        else:
            # Mode B: Standard acrossfade chaining với curve=qsin và apad
            if len(video_paths) == 2:
                filter_parts.append(f"[a0_norm][a1_norm]acrossfade=d={crossfade_dur}:c1=qsin:c2=qsin[a_raw_faded]")
            elif len(video_paths) > 2:
                filter_parts.append(f"[a0_norm][a1_norm]acrossfade=d={crossfade_dur}:c1=qsin:c2=qsin[a_tmp1]")
                for i in range(2, len(video_paths)):
                    in_a = f"[a_tmp{i-1}]"
                    next_a = f"[a{i}_norm]"
                    out_a = "[a_raw_faded]" if i == len(video_paths) - 1 else f"[a_tmp{i}]"
                    filter_parts.append(f"{in_a}{next_a}acrossfade=d={crossfade_dur}:c1=qsin:c2=qsin{out_a}")

            # Đệm đuôi apad để audio khớp tuyệt đối thời lượng video
            filter_parts.append("[a_raw_faded]apad[a_faded]")

        # 3. Chuẩn hóa EBU R128 Loudness
        if normalize_lufs:
            filter_parts.append("[a_faded]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]")
        else:
            filter_parts.append("[a_faded]anull[a_out]")

        filter_complex_str = ";".join(filter_parts)

        cmd = [
            self.ffmpeg, "-y",
            *inputs,
            "-filter_complex", filter_complex_str,
            "-map", "[v_raw]",
            "-map", "[a_out]",
            "-c:v", "libx264",
            "-crf", "18",
            "-preset", "slow",
            "-c:a", "aac",
            "-b:a", "192k",
            "-ar", "48000",
            "-shortest",
            output_path
        ]

        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode == 0:
            print(f"[✓] Ghép video và đồng nhất âm thanh THÀNH CÔNG: {output_path}")
            return True
        else:
            print(f"[!] Lỗi FFmpeg khi ghép:")
            print(res.stderr[-500:])
            return False

    def create_l_cut_bridge(
        self,
        shot_kieu_path: str,
        shot_van_path: str,
        output_path: str,
        l_cut_overlap_sec: float = 3.5,
        lowpass_freq: int = 3500
    ) -> bool:
        """
        Kỹ thuật L-Cut kinh điển: Tiếng đàn tam thập lục của Thúy Kiều tiếp tục
        vọng sang phân cảnh của Thúy Vân bên thềm hiên (lọc tần số lowpass giả lập khoảng cách),
        tạo sự gắn kết không gian và xúc cảm tuyệt đối giữa 2 chị em.
        """
        print(f"\n[*] Đang khởi tạo L-Cut Audio Bridge ({l_cut_overlap_sec}s overlap) giữa Kiều và Vân...")
        cmd = [
            self.ffmpeg, "-y",
            "-i", shot_kieu_path,
            "-i", shot_van_path,
            "-filter_complex",
            # Trích xuất 3.5s cuối tiếng đàn Kiều, áp dụng lowpass và reverb nhẹ
            f"[0:a]atrim=start=6.5:end=10.0,asetpts=PTS-STARTPTS,lowpass=f={lowpass_freq},volume=0.7[kieu_echo];"
            # Hòa âm tiếng đàn vọng vào đầu cảnh Thúy Vân
            f"[1:a]volume=0.9[van_ambient];"
            f"[kieu_echo]adelay=0|0[kieu_delayed];"
            f"[van_ambient][kieu_delayed]amix=inputs=2:duration=first:dropout_transition=2[van_a_mix];"
            # Nối audio Kiều đầy đủ với audio Vân đã mix
            f"[0:a][van_a_mix]acrossfade=d=0.75:c1=qsin:c2=qsin[a_lcut];"
            # Nối video 2 shot
            f"[0:v][1:v]concat=n=2:v=1:a=0[v_out]",
            "-map", "[v_out]",
            "-map", "[a_lcut]",
            "-c:v", "libx264",
            "-crf", "18",
            "-c:a", "aac",
            "-b:a", "192k",
            "-ar", "48000",
            output_path
        ]
        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode == 0:
            print(f"[✓] Đã tạo thành công phân đoạn L-Cut liền mạch: {output_path}")
            return True
        else:
            print("[!] Lỗi tạo L-Cut:")
            print(res.stderr[-400:])
            return False

    def measure_loudness(
        self,
        media_path: str,
        target_lufs: float = -14.0,
        target_tp: float = -1.0,
        target_lra: float = 9.0
    ) -> Optional[Dict]:
        """
        Thực hiện Pass 1 đo kiểm âm lượng EBU R128 của file media qua FFmpeg loudnorm JSON.
        Trả về dictionary chứa: input_i, input_tp, input_lra, input_thresh, target_offset, is_silent.
        Bảo vệ chống lỗi -inf silence guard ngăn ngừa crash FFmpeg.
        """
        if not os.path.exists(media_path):
            return None

        cmd = [
            self.ffmpeg, "-hide_banner", "-y",
            "-i", media_path,
            "-vn", "-sn", "-dn",
            "-af", f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}:print_format=json",
            "-f", "null", "-"
        ]
        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode != 0:
            return None

        matches = re.findall(r'\{[\s\S]*?\}', res.stderr)
        if not matches:
            return None

        try:
            data = json.loads(matches[-1])
            input_i = str(data.get("input_i", "-99.0"))
            input_tp = str(data.get("input_tp", "-99.0"))
            input_thresh = str(data.get("input_thresh", "-99.0"))

            # -inf silence guard: Phát hiện âm thanh câm hoặc quá nhỏ
            if input_i == "-inf" or input_tp == "-inf" or input_thresh == "-inf":
                data["is_silent"] = True
            elif float(input_i) < -90.0:
                data["is_silent"] = True
            else:
                data["is_silent"] = False
            return data
        except Exception:
            return None

    def normalize_loudness(
        self,
        input_video: str,
        output_video: str,
        target_lufs: float = -14.0,
        target_tp: float = -1.0,
        target_lra: float = 9.0,
        two_pass: bool = True
    ) -> bool:
        """
        Chuẩn hóa âm lượng EBU R128 cho video thành phẩm theo chuẩn YouTube Green Dollar.
        Hỗ trợ Two-Pass Linear Normalization (tránh pumping, bảo toàn dynamic range).
        - Target Integrated Loudness: -14.0 LUFS (±0.5 LUFS)
        - True Peak: -1.0 dBTP
        - Loudness Range: 9.0 LU (9.0 - 11.0 LU)
        - Sample Rate: 48000 Hz, Codec AAC 192k
        """
        if not os.path.exists(input_video):
            print(f"[!] Không tìm thấy file đầu vào: {input_video}")
            return False

        print(f"[*] Đang chuẩn hóa âm lượng EBU R128 ({target_lufs} LUFS, Two-Pass={two_pass}) cho: {os.path.basename(input_video)}...")
        os.makedirs(os.path.dirname(os.path.abspath(output_video)), exist_ok=True)

        loudnorm_filter = f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}"

        # Thực thi Pass 1 đo kiểm nếu bật two_pass
        if two_pass:
            meas = self.measure_loudness(input_video, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra)
            if meas and not meas.get("is_silent"):
                try:
                    mi = float(meas.get("input_i", "-24.0"))
                    mt = float(meas.get("input_tp", "-2.0"))
                    ml = float(meas.get("input_lra", "7.0"))
                    mth = float(meas.get("input_thresh", "-34.0"))
                    off = float(meas.get("target_offset", "0.0"))
                    if mi >= -99.0 and mt >= -99.0:
                        loudnorm_filter = (
                            f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}:"
                            f"measured_I={mi:.2f}:measured_TP={mt:.2f}:measured_LRA={ml:.2f}:"
                            f"measured_thresh={mth:.2f}:offset={off:.2f}:linear=true"
                        )
                        print(f"   [Pass 1 Đạt] Measured I={mi:.1f} LUFS, TP={mt:.1f} dBTP, Offset={off:.1f} dB -> Khởi động Pass 2 Linear...")
                except (ValueError, TypeError):
                    print("   [Pass 1 Parse Warning] Giá trị đo kiểm không hợp lệ, fallback single-pass loudnorm.")
            elif meas and meas.get("is_silent"):
                print("   [Pass 1 Silence Guard] File câm (-inf), áp dụng bộ lọc anull an toàn để chống crash AAC encoder.")
                loudnorm_filter = "anull"
            else:
                print("   [Pass 1 Fallback] Không đo được thông số, sử dụng single-pass loudnorm an toàn.")
        else:
            meas = self.measure_loudness(input_video, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra)
            if meas and meas.get("is_silent"):
                print("   [Single Pass Silence Guard] File câm (-inf), áp dụng anull an toàn để chống crash AAC encoder.")
                loudnorm_filter = "anull"

        cmd = [
            self.ffmpeg, "-y",
            "-i", input_video,
            "-af", loudnorm_filter,
            "-c:v", "copy",
            "-c:a", "aac",
            "-b:a", "192k",
            "-ar", "48000",
            output_video
        ]

        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode == 0:
            print(f"[✓] Chuẩn hóa âm lượng thành công: {output_video}")
            return True
        else:
            print("    [!] Stream copy video không tương thích, fallback re-encode video...")
            cmd_fallback = [
                self.ffmpeg, "-y",
                "-i", input_video,
                "-af", loudnorm_filter,
                "-c:v", "libx264", "-crf", "18", "-preset", "slow",
                "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
                output_video
            ]
            res2 = subprocess.run(cmd_fallback, capture_output=True, text=True)
            if res2.returncode == 0:
                print(f"[✓] Chuẩn hóa âm lượng thành công (re-encode): {output_video}")
                return True
            else:
                print(f"[!] Lỗi chuẩn hóa âm lượng FFmpeg: {res2.stderr[-400:]}")
                return False

    def layer_bgm_with_ducking(
        self,
        input_video: str,
        bgm_path: str,
        output_video: str,
        duck_db: float = -14.0,
        bgm_vol: float = 0.35,
        ambience_path: Optional[str] = None,
        ambience_vol: float = 0.35
    ) -> bool:
        """
        Trải thảm nhạc nền (BGM) cho video với kỹ thuật Dynamic Ducking và EBU R128:
        - BGM tự động lặp nếu ngắn hơn video (-stream_loop -1), duration=first.
        - Khi có thoại/foley ở track gốc (0:a), BGM tự hạ âm lượng (ducking).
        - Tùy chọn phối trộn thêm Stem 2 Ambience nếu được cung cấp.
        - Chuẩn hóa đầu ra về -14 LUFS / -1.0 dBTP.
        """
        if not os.path.exists(input_video) or not os.path.exists(bgm_path):
            print(f"[!] File không tồn tại: input={input_video}, bgm={bgm_path}")
            return False

        print(f"[*] Đang trải thảm BGM ({os.path.basename(bgm_path)}) với Sidechain Ducking vào video...")
        os.makedirs(os.path.dirname(os.path.abspath(output_video)), exist_ok=True)

        inputs = ["-i", input_video, "-stream_loop", "-1", "-i", bgm_path]

        if ambience_path and os.path.exists(ambience_path):
            inputs.extend(["-stream_loop", "-1", "-i", ambience_path])
            filter_str = (
                f"[0:a]asplit=3[onset_main][onset_bgm][onset_amb];"
                f"[1:a]volume={bgm_vol},equalizer=f=2150:width_type=h:width=2700:g=-3.5[bgm_base];"
                f"[2:a]volume={ambience_vol}[amb_base];"
                f"[bgm_base][onset_bgm]sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[bgm_ducked];"
                f"[amb_base][onset_amb]sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[amb_ducked];"
                f"[onset_main][bgm_ducked][amb_ducked]amix=inputs=3:duration=first:dropout_transition=2[a_mixed];"
                f"[a_mixed]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]"
            )
        else:
            filter_str = (
                f"[0:a]asplit=2[onset_main][onset_bgm];"
                f"[1:a]volume={bgm_vol}[bgm_base];"
                f"[bgm_base][onset_bgm]sidechaincompress=threshold=0.08:ratio=4:attack=20:release=350[bgm_ducked];"
                f"[onset_main][bgm_ducked]amix=inputs=2:duration=first:dropout_transition=2[a_mixed];"
                f"[a_mixed]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]"
            )

        cmd = [
            self.ffmpeg, "-y",
            *inputs,
            "-filter_complex", filter_str,
            "-map", "0:v:0",
            "-map", "[a_out]",
            "-c:v", "copy",
            "-c:a", "aac",
            "-b:a", "192k",
            "-ar", "48000",
            "-shortest",
            output_video
        ]

        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode == 0:
            print(f"[✓] Trải thảm BGM và Master âm thanh thành công: {output_video}")
            return True
        else:
            print("    [!] Thử lại với re-encode video...")
            cmd_fallback = [
                self.ffmpeg, "-y",
                *inputs,
                "-filter_complex", filter_str,
                "-map", "0:v:0",
                "-map", "[a_out]",
                "-c:v", "libx264", "-crf", "18", "-preset", "slow",
                "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
                "-shortest",
                output_video
            ]
            res2 = subprocess.run(cmd_fallback, capture_output=True, text=True)
            if res2.returncode == 0:
                print(f"[✓] Trải thảm BGM thành công (re-encode): {output_video}")
                return True
            else:
                print(f"[!] Lỗi trải BGM FFmpeg: {res2.stderr[-400:]}")
                return False

    def mix_four_stems(
        self,
        video_path: str,
        output_path: str,
        stem1_bgm: Optional[str] = None,
        stem2_ambience: Optional[str] = None,
        stem3_foley: Optional[str] = None,
        stem4_dialogue: Optional[str] = None,
        bgm_vol: float = 0.35,
        ambience_vol: float = 0.35,
        foley_vol: float = 0.80,
        dialogue_vol: float = 1.0,
        duck_db: float = -14.0,
        target_lufs: float = -14.0,
        target_tp: float = -1.0,
        target_lra: float = 9.0,
        two_pass: bool = True
    ) -> bool:
        """
        Phối âm toàn diện theo Kiến Trúc 4 Stems Độc Lập chuẩn Hollywood/EBU R128:
        - Stem 1 (BGM): Lặp liên tục (-stream_loop -1), EQ notch 800Hz - 3500Hz, ducking khi có thoại.
        - Stem 2 (Ambience): Âm cảnh môi trường 3D liên tục, ducking khi có thoại.
        - Stem 3 (Foley): Âm thanh tác động cơ học sắc nét (áo lụa, trâm cài, tách trà, vó ngựa).
        - Stem 4 (Dialogue): Thoại & ngâm thơ, kích hoạt sidechain ducking (-14dB) cho Stem 1 & Stem 2.
        - Master Output: Chuẩn hóa Two-Pass Linear EBU R128 (-14 LUFS, TP -1.0, LRA 9-11, 48kHz AAC).
        """
        if not os.path.exists(video_path):
            print(f"[!] Không tìm thấy video đầu vào: {video_path}")
            return False

        os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True)
        print(f"\n🎛️ BẮT ĐẦU PHỐI ÂM 4 STEMS CHO: {os.path.basename(video_path)}")

        inputs = ["-i", video_path]
        filter_parts = []
        mix_inputs = []
        input_idx = 1

        # Xác định các stem cần ducking
        has_bgm = bool(stem1_bgm and os.path.exists(stem1_bgm))
        has_amb = bool(stem2_ambience and os.path.exists(stem2_ambience))
        duck_targets = (1 if has_bgm else 0) + (1 if has_amb else 0)

        # Xác định track thoại (Stem 4 hoặc audio gốc của video)
        has_dialogue_stem = bool(stem4_dialogue and os.path.exists(stem4_dialogue))
        has_video_audio = self.inspect_shot_audio(video_path).get("has_audio", False)

        side_bgm_tag = None
        side_amb_tag = None

        if has_dialogue_stem:
            inputs.extend(["-i", stem4_dialogue])
            diag_idx = input_idx
            input_idx += 1
            filter_parts.append(f"[{diag_idx}:a]aresample=48000,aformat=channel_layouts=stereo,volume={dialogue_vol}[stem4_voc]")
            if duck_targets == 2:
                filter_parts.append("[stem4_voc]asplit=3[voc_main][voc_side1][voc_side2]")
                side_bgm_tag = "[voc_side1]"
                side_amb_tag = "[voc_side2]"
            elif duck_targets == 1:
                filter_parts.append("[stem4_voc]asplit=2[voc_main][voc_side1]")
                side_bgm_tag = "[voc_side1]"
                side_amb_tag = "[voc_side1]"
            else:
                filter_parts.append("[stem4_voc]anull[voc_main]")
            mix_inputs.append("[voc_main]")
        elif has_video_audio:
            filter_parts.append(f"[0:a]aresample=48000,aformat=channel_layouts=stereo,volume={dialogue_vol}[onset_voc]")
            if duck_targets == 2:
                filter_parts.append("[onset_voc]asplit=3[voc_main][voc_side1][voc_side2]")
                side_bgm_tag = "[voc_side1]"
                side_amb_tag = "[voc_side2]"
            elif duck_targets == 1:
                filter_parts.append("[onset_voc]asplit=2[voc_main][voc_side1]")
                side_bgm_tag = "[voc_side1]"
                side_amb_tag = "[voc_side1]"
            else:
                filter_parts.append("[onset_voc]anull[voc_main]")
            mix_inputs.append("[voc_main]")

        # Xử lý Stem 1: BGM (kèm notch filter 800Hz - 3500Hz)
        if has_bgm:
            inputs.extend(["-stream_loop", "-1", "-i", stem1_bgm])
            bgm_idx = input_idx
            input_idx += 1
            filter_parts.append(
                f"[{bgm_idx}:a]aresample=48000,aformat=channel_layouts=stereo,volume={bgm_vol},"
                f"equalizer=f=2150:width_type=h:width=2700:g=-3.5[bgm_base]"
            )
            if side_bgm_tag:
                filter_parts.append(f"[bgm_base]{side_bgm_tag}sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[bgm_ducked]")
                mix_inputs.append("[bgm_ducked]")
            else:
                mix_inputs.append("[bgm_base]")

        # Xử lý Stem 2: Ambience
        if has_amb:
            inputs.extend(["-stream_loop", "-1", "-i", stem2_ambience])
            amb_idx = input_idx
            input_idx += 1
            filter_parts.append(f"[{amb_idx}:a]aresample=48000,aformat=channel_layouts=stereo,volume={ambience_vol}[amb_base]")
            if side_amb_tag:
                filter_parts.append(f"[amb_base]{side_amb_tag}sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[amb_ducked]")
                mix_inputs.append("[amb_ducked]")
            else:
                mix_inputs.append("[amb_base]")

        # Xử lý Stem 3: Foley
        if stem3_foley and os.path.exists(stem3_foley):
            inputs.extend(["-i", stem3_foley])
            fol_idx = input_idx
            input_idx += 1
            filter_parts.append(f"[{fol_idx}:a]aresample=48000,aformat=channel_layouts=stereo,volume={foley_vol}[fol_base]")
            mix_inputs.append("[fol_base]")

        # Nếu không có thêm stem nào, chỉ cần normalize loudness
        if not mix_inputs:
            return self.normalize_loudness(video_path, output_path, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra, two_pass=two_pass)

        # Trộn tất cả active stems
        amix_line = f"{''.join(mix_inputs)}amix=inputs={len(mix_inputs)}:duration=first:dropout_transition=2[a_mix]"
        filter_parts.append(amix_line)
        filter_parts.append("[a_mix]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]")
        filter_complex_str = ";".join(filter_parts)

        cmd = [
            self.ffmpeg, "-y",
            *inputs,
            "-filter_complex", filter_complex_str,
            "-map", "0:v:0",
            "-map", "[a_out]",
            "-c:v", "copy",
            "-c:a", "aac",
            "-b:a", "192k",
            "-ar", "48000",
            "-shortest",
            output_path
        ]

        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode == 0:
            print(f"[✓] Phối 4 stems thành công: {output_path}")
            if two_pass:
                # Chạy Pass 2 linear hoàn thiện trên master output
                temp_master = output_path + ".temp_norm.mp4"
                if os.path.exists(output_path):
                    shutil.move(output_path, temp_master)
                    ok = self.normalize_loudness(temp_master, output_path, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra, two_pass=True)
                    if os.path.exists(temp_master):
                        os.unlink(temp_master)
                    return ok
            return True
        else:
            print("    [!] Thử lại với re-encode video...")
            cmd_fallback = [
                self.ffmpeg, "-y",
                *inputs,
                "-filter_complex", filter_complex_str,
                "-map", "0:v:0",
                "-map", "[a_out]",
                "-c:v", "libx264", "-crf", "18", "-preset", "slow",
                "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
                "-shortest",
                output_path
            ]
            res2 = subprocess.run(cmd_fallback, capture_output=True, text=True)
            if res2.returncode == 0 and two_pass:
                temp_master = output_path + ".temp_norm.mp4"
                if os.path.exists(output_path):
                    shutil.move(output_path, temp_master)
                    ok = self.normalize_loudness(temp_master, output_path, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra, two_pass=True)
                    if os.path.exists(temp_master):
                        os.unlink(temp_master)
                    return ok
            return res2.returncode == 0

    def smart_trim_shot_head(
        self,
        video_path: str,
        output_path: Optional[str] = None,
        trim_dur: float = 1.0,
        detect_static: bool = True
    ) -> str:
        """
        Cắt tỉa thông minh phần đầu cú máy (Dynamic Head Trim):
        - Kiểm tra thời lượng: chỉ thực hiện khi video đủ dài (duration > trim_dur + 0.5s).
        - Tự động phát hiện hiện tượng đơ hình/hitch tĩnh trong 1 giây đầu (24 frames) của AI video.
        - Nếu có đơ hình hoặc detect_static=False: áp dụng -ss {trim_dur} để loại bỏ frame tĩnh đầu.
        - Bảo toàn 100% âm thanh AAC tại 48kHz stereo.
        """
        if not os.path.exists(video_path):
            return video_path

        # Kiểm tra thời lượng tổng thể để tránh cắt trụi video ngắn
        info = self.inspect_shot_audio(video_path)
        total_dur = float(info.get("duration", 0.0))
        if total_dur > 0.0 and total_dur <= (trim_dur + 0.5):
            # Video quá ngắn để trim 1.0s an toàn
            return video_path

        should_trim = not detect_static
        if detect_static:
            try:
                import cv2
                import numpy as np
                cap = cv2.VideoCapture(video_path)
                if cap.isOpened():
                    fps = cap.get(cv2.CAP_PROP_FPS) or 24.0
                    check_idx = max(1, int(fps * min(trim_dur * 0.8, 0.8)))
                    ret1, f1 = cap.read()
                    cap.set(cv2.CAP_PROP_POS_FRAMES, check_idx)
                    ret2, f2 = cap.read()
                    cap.release()
                    if ret1 and ret2 and f1 is not None and f2 is not None:
                        diff = float(np.mean(np.abs(f1.astype(float) - f2.astype(float))))
                        if diff < 1.5:  # Khung hình gần như bất động trong khoảng đầu
                            should_trim = True
                            print(f"      ✂️ [SMART TRIM] Phát hiện đầu video tĩnh (diff={diff:.2f} < 1.5 tại frame {check_idx}). Kích hoạt -ss {trim_dur:.1f}s.")
            except Exception:
                should_trim = False

        if not should_trim:
            return video_path

        out = output_path or str(Path(video_path).with_name(f"{Path(video_path).stem}_trimmed.mp4"))
        os.makedirs(os.path.dirname(os.path.abspath(out)), exist_ok=True)

        cmd = [
            self.ffmpeg, "-y",
            "-ss", str(trim_dur),
            "-i", video_path,
            "-c:v", "libx264", "-crf", "18", "-preset", "fast",
            "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
            out
        ]
        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode == 0 and os.path.exists(out) and os.path.getsize(out) > 1000:
            return out
        return video_path

    def fit_footage_to_narration(
        self,
        video_path: str,
        narration_audio_path: str,
        output_path: str,
        buffer_sec: float = 0.3
    ) -> bool:
        """
        Khớp thời lượng video với âm thanh thuyết minh (Narrator Fitting):
        - Đo đạc chính xác thời lượng giọng đọc qua ffprobe.
        - Khớp video theo độ dài thuyết minh + buffer an toàn (buffer_sec).
        - Nếu video ngắn hơn thời lượng thuyết minh, tự động giữ frame cuối (tpad clone) để tránh đứt video sớm.
        - Trộn âm thanh thuyết minh với âm thanh foley/ambience gốc.
        - Chuẩn hóa đầu ra đạt chuẩn EBU R128 (-14 LUFS).
        """
        if not os.path.exists(video_path) or not os.path.exists(narration_audio_path):
            print(f"[!] File video hoặc narration không tồn tại: {video_path}, {narration_audio_path}")
            return False

        # Đo thời lượng narration
        probe_cmd = [
            self.ffprobe, "-v", "error",
            "-show_entries", "format=duration",
            "-of", "json", narration_audio_path
        ]
        res_a = subprocess.run(probe_cmd, capture_output=True, text=True)
        narr_dur = 0.0
        if res_a.returncode == 0:
            try:
                narr_dur = float(json.loads(res_a.stdout).get("format", {}).get("duration", 0.0))
            except Exception:
                pass

        if narr_dur <= 0.0:
            print("[!] Không thể xác định thời lượng file thuyết minh.")
            return False

        target_dur = narr_dur + buffer_sec
        os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True)
        temp_out = output_path + ".temp_fit.mp4"

        # Trộn narration và video, hỗ trợ kéo dài video bằng tpad nếu video ngắn hơn target_dur
        v_filter = "[0:v]tpad=stop_mode=clone:stop=-1[v_pad]"
        filter_str = (
            f"{v_filter};"
            f"[1:a]aresample=48000,aformat=sample_rates=48000:channel_layouts=stereo,volume=1.0[narr];"
            f"[0:a]aresample=48000,aformat=sample_rates=48000:channel_layouts=stereo,volume=0.3[bg_audio];"
            f"[narr][bg_audio]amix=inputs=2:duration=longest:dropout_transition=2[a_mix]"
        )

        cmd = [
            self.ffmpeg, "-y",
            "-i", video_path,
            "-i", narration_audio_path,
            "-t", f"{target_dur:.3f}",
            "-filter_complex", filter_str,
            "-map", "[v_pad]",
            "-map", "[a_mix]",
            "-c:v", "libx264", "-crf", "18", "-preset", "fast",
            "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
            temp_out
        ]
        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode != 0 or not os.path.exists(temp_out):
            # Fallback nếu video gốc không có luồng audio
            filter_fb = f"[0:v]tpad=stop_mode=clone:stop=-1[v_pad];[1:a]aresample=48000,aformat=sample_rates=48000:channel_layouts=stereo[a_pad]"
            cmd_fallback = [
                self.ffmpeg, "-y",
                "-i", video_path,
                "-i", narration_audio_path,
                "-t", f"{target_dur:.3f}",
                "-filter_complex", filter_fb,
                "-map", "[v_pad]",
                "-map", "[a_pad]",
                "-c:v", "libx264", "-crf", "18", "-preset", "fast",
                "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
                temp_out
            ]
            res_fb = subprocess.run(cmd_fallback, capture_output=True, text=True)
            if res_fb.returncode != 0 or not os.path.exists(temp_out):
                return False

        # Chuẩn hóa âm lượng EBU R128 (-14 LUFS)
        norm_ok = self.normalize_loudness(temp_out, output_path, target_lufs=-14.0, two_pass=True)
        if os.path.exists(temp_out):
            try:
                os.unlink(temp_out)
            except Exception:
                pass
        return norm_ok

    def stitch_flowkit_smart(
        self,
        shot_entries: List[Dict[str, Any]],
        output_path: str,
        xfade_dur: float = 0.5,
        normalize_lufs: bool = True
    ) -> bool:
        """
        Ghép nối phân cảnh thông minh theo chuẩn FlowKit:
        - Giữa các shot cùng nhân vật/nối tiếp (CONTINUOUS_TAKE): Áp dụng xfade 0.5s cho video
          VÀ acrossfade 0.5s cho audio đồng bộ chính xác đến từng millisecond (0ms drift).
        - Giữa các shot chuyển vai/đổi cảnh (CINEMATIC_CUT): Giữ nguyên hard cut dứt khoát
          kết hợp micro-fade 30ms triệt tiêu tiếng click/pop.
        - Hỗ trợ bất kỳ số lượng shot nào (N >= 1) với tổ hợp ngẫu nhiên các loại take.
        - Chuẩn hóa toàn bộ video xuất xưởng về mốc -14.0 LUFS EBU R128, 48kHz stereo.
        """
        if not shot_entries:
            return False

        valid_entries = [e for e in shot_entries if os.path.exists(e.get("video_path", ""))]
        if not valid_entries:
            print("[!] Không có video hợp lệ nào để ghép cảnh.")
            return False

        # HARD GATE: Khóa cứng tỷ lệ 16:9 Cinema Wide, cấm tuyệt đối video dọc (9:16)
        for e in valid_entries:
            vp = e.get("video_path", "")
            w, h = self.inspect_video_dimensions(vp)
            if w > 0 and h > 0 and (w / h < 1.33):
                raise ValueError(
                    f"[CRITICAL ASPECT RATIO REJECTION] Video {os.path.basename(vp)} có tỷ lệ dọc {w}x{h} (ratio={w/h:.3f}). "
                    f"Bản Master điện ảnh 16:9 NGHIÊM CẤM video 9:16. Ghép nối bị hủy! Cần render lại shot 16:9."
                )

        if len(valid_entries) == 1:
            v_in = valid_entries[0]["video_path"]
            return self.normalize_loudness(v_in, output_path, target_lufs=-14.0, two_pass=True)

        os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True)
        temp_stitched = output_path + ".temp_flowkit.mp4"

        n = len(valid_entries)
        inputs = []
        for e in valid_entries:
            inputs.extend(["-i", e["video_path"]])

        reports = [self.inspect_shot_audio(e["video_path"]) for e in valid_entries]

        # 1. Tiền xử lý chuẩn hóa video và audio stream cho từng input
        filter_parts = []
        for i, rep in enumerate(reports):
            # Chuẩn hóa video stream về chuẩn 1280x720, 24fps, SAR 1:1
            filter_parts.append(
                f"[{i}:v]scale=1280:720:force_original_aspect_ratio=decrease,pad=1280:720:(ow-iw)/2:(oh-ih)/2,setsar=1,fps=24[v{i}_norm]"
            )
            if rep["has_audio"]:
                filter_parts.append(
                    f"[{i}:a]aresample=48000,aformat=sample_rates=48000:channel_layouts=stereo[a{i}_norm]"
                )
            else:
                cdur = rep.get("duration", 10.0)
                if cdur <= 0:
                    cdur = 10.0
                filter_parts.append(f"aevalsrc=0:d={cdur}:s=48000:c=stereo[a{i}_norm]")

        # 2. Gom nhóm các shot liền kề có take_type == CONTINUOUS_TAKE
        # Một nhóm mới bắt đầu tại index 0, hoặc tại index i mà valid_entries[i].take_type != CONTINUOUS_TAKE
        groups: List[List[int]] = []
        current_group: List[int] = [0]
        for i in range(1, n):
            tt = valid_entries[i].get("take_type", "CINEMATIC_CUT")
            if tt == "CONTINUOUS_TAKE":
                current_group.append(i)
            else:
                groups.append(current_group)
                current_group = [i]
        if current_group:
            groups.append(current_group)

        # 3. Với mỗi nhóm, nếu có > 1 shot: nối bằng xfade (video) và acrossfade (audio)
        group_video_tags = []
        group_audio_tags = []
        group_durations = []

        for g_idx, grp in enumerate(groups):
            if len(grp) == 1:
                idx = grp[0]
                group_video_tags.append(f"[v{idx}_norm]")
                group_audio_tags.append(f"[a{idx}_norm]")
                group_durations.append(float(reports[idx].get("duration", 10.0)))
            else:
                # Chaining xfade & acrossfade trong nội bộ nhóm grp
                curr_v = f"[v{grp[0]}_norm]"
                curr_a = f"[a{grp[0]}_norm]"
                cum_dur = float(reports[grp[0]].get("duration", 10.0))

                for step_i, next_idx in enumerate(grp[1:], 1):
                    next_dur = float(reports[next_idx].get("duration", 10.0))
                    offset = max(0.0, cum_dur - xfade_dur)
                    out_v = f"[v_g{g_idx}_{step_i}]"
                    out_a = f"[a_g{g_idx}_{step_i}]"

                    # Video xfade
                    filter_parts.append(
                        f"{curr_v}[v{next_idx}_norm]xfade=transition=fade:duration={xfade_dur:.2f}:offset={offset:.2f}{out_v}"
                    )
                    # Audio acrossfade (khớp chính xác thời lượng và transition với video)
                    filter_parts.append(
                        f"{curr_a}[a{next_idx}_norm]acrossfade=d={xfade_dur:.2f}:c1=qsin:c2=qsin{out_a}"
                    )

                    curr_v = out_v
                    curr_a = out_a
                    cum_dur = offset + next_dur

                group_video_tags.append(curr_v)
                group_audio_tags.append(curr_a)
                group_durations.append(cum_dur)

        # 4. Ghép giữa các groups bằng Hard Cut (CINEMATIC_CUT)
        m = len(groups)
        if m == 1:
            v_master = group_video_tags[0]
            a_master = group_audio_tags[0]
        else:
            # Video concat giữa các groups
            v_grp_inputs = "".join(group_video_tags)
            filter_parts.append(f"{v_grp_inputs}concat=n={m}:v=1:a=0[v_master]")
            v_master = "[v_master]"

            # Audio concat giữa các groups có micro-fade 30ms để tránh click/pop
            micro_dur = 0.030
            faded_a_tags = []
            for j in range(m):
                g_dur = group_durations[j]
                out_start = max(0.0, g_dur - micro_dur)
                tag_out = f"[a_grp_{j}_faded]"
                filter_parts.append(
                    f"{group_audio_tags[j]}afade=t=in:st=0:d={micro_dur}:curve=qsin,"
                    f"afade=t=out:st={out_start:.3f}:d={micro_dur}:curve=qsin{tag_out}"
                )
                faded_a_tags.append(tag_out)

            a_grp_inputs = "".join(faded_a_tags)
            filter_parts.append(f"{a_grp_inputs}concat=n={m}:v=0:a=1[a_master]")
            a_master = "[a_master]"

        filter_complex_str = ";".join(filter_parts)

        cmd = [
            self.ffmpeg, "-y",
            *inputs,
            "-filter_complex", filter_complex_str,
            "-map", v_master,
            "-map", a_master,
            "-c:v", "libx264", "-crf", "18", "-preset", "fast",
            "-c:a", "aac", "-b:a", "192k", "-ar", "48000",
            temp_stitched
        ]

        res = subprocess.run(cmd, capture_output=True, text=True)
        if res.returncode != 0 or not os.path.exists(temp_stitched):
            # Fallback về stitch_with_audio_crossfade thông thường nếu có lỗi filter complex
            video_paths = [e["video_path"] for e in valid_entries]
            return self.stitch_with_audio_crossfade(video_paths, output_path, mode="boundary_smoothing")

        if normalize_lufs:
            norm_ok = self.normalize_loudness(temp_stitched, output_path, target_lufs=-14.0, two_pass=True)
            if os.path.exists(temp_stitched):
                try:
                    os.unlink(temp_stitched)
                except Exception:
                    pass
            return norm_ok

        shutil.move(temp_stitched, output_path)
        return True


if __name__ == "__main__":
    parser = argparse.ArgumentParser(description="KieuStory Audio Continuity & Soundscape Engine (4-Stem & EBU R128)")
    parser.add_argument("--inspect", nargs="+", help="Danh sách video shot cần kiểm tra âm thanh")
    parser.add_argument("--stitch", nargs="+", help="Ghép nối danh sách video với crossfade âm thanh")
    parser.add_argument("--output", default="04_Assets/videos/master_audio_seamless.mp4", help="Đường dẫn xuất")
    parser.add_argument("--crossfade", type=float, default=1.0, help="Thời gian crossfade âm thanh (giây)")
    parser.add_argument("--mode", type=str, default="acrossfade", choices=["acrossfade", "boundary_smoothing", "micro_crossfade"], help="Chế độ chuyển tiếp âm thanh")
    parser.add_argument("--l-cut", nargs=2, help="Tạo L-Cut giữa 2 video (Shot Kiều và Shot Vân)")
    parser.add_argument("--normalize", type=str, help="Chuẩn hóa âm lượng EBU R128 (-14 LUFS) Two-Pass cho 1 video")
    parser.add_argument("--measure", type=str, help="Đo kiểm âm lượng Pass 1 JSON qua FFmpeg loudnorm")
    parser.add_argument("--mix-stems", action="store_true", help="Kích hoạt bộ trộn 4 Stems")
    parser.add_argument("--video", type=str, help="Video nguồn cho mix-stems")
    parser.add_argument("--bgm", type=str, help="Stem 1: BGM audio path")
    parser.add_argument("--ambience", type=str, help="Stem 2: Ambience audio path")
    parser.add_argument("--foley", type=str, help="Stem 3: Foley audio path")
    parser.add_argument("--dialogue", type=str, help="Stem 4: Dialogue audio path")

    args = parser.parse_args()
    engine = AudioContinuityEngine()

    if args.inspect:
        engine.inspect_sequence(args.inspect)
    elif args.stitch:
        engine.stitch_with_audio_crossfade(args.stitch, args.output, crossfade_dur=args.crossfade, mode=args.mode)
    elif args.l_cut:
        engine.create_l_cut_bridge(args.l_cut[0], args.l_cut[1], args.output)
    elif args.normalize:
        engine.normalize_loudness(args.normalize, args.output, target_lufs=-14.0, two_pass=True)
    elif args.measure:
        m = engine.measure_loudness(args.measure)
        print(json.dumps(m, indent=2))
    elif args.mix_stems and args.video:
        engine.mix_four_stems(
            args.video, args.output,
            stem1_bgm=args.bgm,
            stem2_ambience=args.ambience,
            stem3_foley=args.foley,
            stem4_dialogue=args.dialogue
        )
    else:
        # Mặc định: Kiểm tra chuỗi video Scene 1
        scene1_shots = [
            r"04_Assets\videos\ep01_scene01_shot01_10s.mp4",
            r"04_Assets\videos\ep01_scene01_shot02_10s.mp4",
            r"04_Assets\videos\ep01_scene01_shot03_10s.mp4",
            r"04_Assets\videos\ep01_scene01_shot04_10s.mp4",
            r"04_Assets\videos\ep01_scene01_shot05_10s.mp4"
        ]
        engine.inspect_sequence(scene1_shots)
