--- a/05_Production_Pipeline/audio_continuity_engine.py +++ b/05_Production_Pipeline/audio_continuity_engine.py @@ -1,25 +1,34 @@ """ ============================================================================= -KIEU STORY AI CINEMA - AUDIO CONTINUITY & SOUNDSCAPE ENGINE +KIEU STORY AI CINEMA - AUDIO CONTINUITY & SOUNDSCAPE ENGINE (UPGRADED M3) ============================================================================= Công cụ điều phối và chuẩn hóa âm thanh đồng nhất giữa các shot video AI (Muse.ai) +- 4-Stem Audio Architecture (Stem 1: BGM, Stem 2: Ambience, Stem 3: Foley, Stem 4: Dialogue). - Equal-Power Audio Crossfade (qsin/cbrt) triệt tiêu tiếng click/pop ngắt cụt. - Xử lý gối đầu âm thanh qua đoạn nối hình ảnh (Transition Audio Bridge). - Kỹ thuật J-Cut / L-Cut không gian (Acoustic Distance Filter) cho tiếng đàn Kiều. -- Chuẩn hóa âm lượng EBU R128 (-14 LUFS) đạt chuẩn phát sóng YouTube Green Dollar. +- Dynamic Sidechain Ducking (hạ BGM & Ambience -14dB khi có thoại / ngâm thơ). +- Chuẩn hóa âm lượng EBU R128 Two-Pass Linear (-14 LUFS, TP -1.0 dBTP, LRA 9-11 LU) đạt chuẩn phát sóng YouTube Green Dollar. ============================================================================= """ import os import sys +import re import json import argparse import subprocess -from typing import List, Dict, Optional +import tempfile +from pathlib import Path +from typing import List, Dict, Optional, Tuple # Khởi tạo UTF-8 cho Windows Console if sys.platform == "win32": - sys.stdout.reconfigure(encoding="utf-8") + try: + sys.stdout.reconfigure(encoding="utf-8") + sys.stderr.reconfigure(encoding="utf-8") + except Exception: + pass # Tìm đường dẫn FFmpeg trong hệ thống DEFAULT_FFMPEG_PATHS = [ @@ -56,6 +65,12 @@ class AudioContinuityEngine: + # 4-Stem Definitions theo Project Bible CINEMATIC_AUDIO_PIPELINE.md + STEM_1_BGM = "Stem 1: BGM & Score" + STEM_2_AMBIENCE = "Stem 2: Ambience & Soundscape" + STEM_3_FOLEY = "Stem 3: Foley & Spot SFX" + STEM_4_DIALOGUE = "Stem 4: Dialogue & Voice-Over" + def __init__(self): self.ffmpeg = get_ffmpeg() self.ffprobe = get_ffprobe() @@ -63,7 +78,7 @@ def inspect_shot_audio(self, video_path: str) -> Dict: """Phân tích chi tiết luồng âm thanh và mức âm lượng của một shot video.""" if not os.path.exists(video_path): - return {"file": video_path, "has_audio": False, "error": "File not found"} + return {"file": video_path, "has_audio": False, "max_volume_db": -99.0, "mean_volume_db": -99.0, "error": "File not found"} # Probe streams probe_cmd = [ @@ -159,8 +174,6 @@ print(f"\n[*] Đang ghép {len(video_paths)} shot với Audio Crossfade {crossfade_dur}s...") os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True) - # Xử lý trường hợp 2 file hoặc nhiều file - # Tạo filter_complex an toàn cho FFmpeg inputs = [] for p in video_paths: inputs.extend(["-i", p]) @@ -261,31 +274,97 @@ print(res.stderr[-400:]) return False + def measure_loudness( + self, + media_path: str, + target_lufs: float = -14.0, + target_tp: float = -1.0, + target_lra: float = 9.0 + ) -> Optional[Dict]: + """ + Thực hiện Pass 1 đo kiểm âm lượng EBU R128 của file media qua FFmpeg loudnorm JSON. + Trả về dictionary chứa: input_i, input_tp, input_lra, input_thresh, target_offset, is_silent. + Trả về None nếu file không tồn tại hoặc lỗi đo kiểm. + """ + if not os.path.exists(media_path): + return None + + cmd = [ + self.ffmpeg, "-hide_banner", "-y", + "-i", media_path, + "-vn", "-sn", "-dn", + "-af", f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}:print_format=json", + "-f", "null", "-" + ] + res = subprocess.run(cmd, capture_output=True, text=True) + if res.returncode != 0: + return None + + # Trích xuất block JSON từ stderr của FFmpeg + matches = re.findall(r'\{[\s\S]*?\}', res.stderr) + if not matches: + return None + + try: + data = json.loads(matches[-1]) + input_i = str(data.get("input_i", "-99.0")) + # Phát hiện âm thanh câm (pure silence hoặc quá nhỏ dưới -99 dB) + if input_i == "-inf" or float(input_i) < -99.0: + data["is_silent"] = True + else: + data["is_silent"] = False + return data + except Exception: + return None + def normalize_loudness( self, input_video: str, output_video: str, target_lufs: float = -14.0, target_tp: float = -1.0, - target_lra: float = 9.0 + target_lra: float = 9.0, + two_pass: bool = True ) -> bool: """ Chuẩn hóa âm lượng EBU R128 cho video thành phẩm theo chuẩn YouTube Green Dollar. - - Target Integrated Loudness: -14.0 LUFS + Hỗ trợ Two-Pass Linear Normalization (tránh pumping, bảo toàn dynamic range). + - Target Integrated Loudness: -14.0 LUFS (±0.5 LUFS) - True Peak: -1.0 dBTP - - Loudness Range: 9.0 LU + - Loudness Range: 9.0 LU (9.0 - 11.0 LU) + - Sample Rate: 48000 Hz, Codec AAC 192k """ if not os.path.exists(input_video): print(f"[!] Không tìm thấy file đầu vào: {input_video}") return False - print(f"[*] Đang chuẩn hóa âm lượng EBU R128 ({target_lufs} LUFS) cho: {os.path.basename(input_video)}...") + print(f"[*] Đang chuẩn hóa âm lượng EBU R128 ({target_lufs} LUFS, Two-Pass={two_pass}) cho: {os.path.basename(input_video)}...") os.makedirs(os.path.dirname(os.path.abspath(output_video)), exist_ok=True) + + loudnorm_filter = f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}" + + # Thực thi Pass 1 đo kiểm nếu bật two_pass + if two_pass: + meas = self.measure_loudness(input_video, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra) + if meas and not meas.get("is_silent"): + mi = meas.get("input_i", "-24.0") + mt = meas.get("input_tp", "-2.0") + ml = meas.get("input_lra", "7.0") + mth = meas.get("input_thresh", "-34.0") + off = meas.get("target_offset", "0.0") + loudnorm_filter = ( + f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}:" + f"measured_I={mi}:measured_TP={mt}:measured_LRA={ml}:" + f"measured_thresh={mth}:offset={off}:linear=true" + ) + print(f" [Pass 1 Đạt] Measured I={mi} LUFS, TP={mt} dBTP, Offset={off} dB -> Khởi động Pass 2 Linear...") + else: + print(" [Pass 1 Fallback] File câm hoặc không đo được thông số, sử dụng single-pass loudnorm an toàn.") cmd = [ self.ffmpeg, "-y", "-i", input_video, - "-af", f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}", + "-af", loudnorm_filter, "-c:v", "copy", "-c:a", "aac", "-b:a", "192k", @@ -302,7 +381,7 @@ cmd_fallback = [ self.ffmpeg, "-y", "-i", input_video, - "-af", f"loudnorm=I={target_lufs}:TP={target_tp}:LRA={target_lra}", + "-af", loudnorm_filter, "-c:v", "libx264", "-crf", "18", "-preset", "slow", "-c:a", "aac", "-b:a", "192k", "-ar", "48000", output_video @@ -321,12 +400,15 @@ bgm_path: str, output_video: str, duck_db: float = -14.0, - bgm_vol: float = 0.35 + bgm_vol: float = 0.35, + ambience_path: Optional[str] = None, + ambience_vol: float = 0.35 ) -> bool: """ Trải thảm nhạc nền (BGM) cho video với kỹ thuật Dynamic Ducking và EBU R128: - - BGM tự động lặp nếu ngắn hơn video, hoặc cắt ngắn theo video (duration=first). + - BGM tự động lặp nếu ngắn hơn video (-stream_loop -1), duration=first. - Khi có thoại/foley ở track gốc (0:a), BGM tự hạ âm lượng (ducking). + - Tùy chọn phối trộn thêm Stem 2 Ambience nếu được cung cấp. - Chuẩn hóa đầu ra về -14 LUFS / -1.0 dBTP. """ if not os.path.exists(input_video) or not os.path.exists(bgm_path): @@ -336,17 +418,29 @@ print(f"[*] Đang trải thảm BGM ({os.path.basename(bgm_path)}) với Sidechain Ducking vào video...") os.makedirs(os.path.dirname(os.path.abspath(output_video)), exist_ok=True) - filter_str = ( - f"[1:a]volume={bgm_vol}[bgm_base];" - f"[bgm_base][0:a]sidechaincompress=threshold=0.08:ratio=4:attack=20:release=350[bgm_ducked];" - f"[0:a][bgm_ducked]amix=inputs=2:duration=first:dropout_transition=2[a_mixed];" - f"[a_mixed]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]" - ) + inputs = ["-i", input_video, "-stream_loop", "-1", "-i", bgm_path] + + if ambience_path and os.path.exists(ambience_path): + inputs.extend(["-stream_loop", "-1", "-i", ambience_path]) + filter_str = ( + f"[1:a]volume={bgm_vol},equalizer=f=2150:width_type=h:width=2700:g=-3.5[bgm_base];" + f"[2:a]volume={ambience_vol}[amb_base];" + f"[bgm_base][0:a]sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[bgm_ducked];" + f"[amb_base][0:a]sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[amb_ducked];" + f"[0:a][bgm_ducked][amb_ducked]amix=inputs=3:duration=first:dropout_transition=2[a_mixed];" + f"[a_mixed]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]" + ) + else: + filter_str = ( + f"[1:a]volume={bgm_vol}[bgm_base];" + f"[bgm_base][0:a]sidechaincompress=threshold=0.08:ratio=4:attack=20:release=350[bgm_ducked];" + f"[0:a][bgm_ducked]amix=inputs=2:duration=first:dropout_transition=2[a_mixed];" + f"[a_mixed]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]" + ) cmd = [ self.ffmpeg, "-y", - "-i", input_video, - "-stream_loop", "-1", "-i", bgm_path, + *inputs, "-filter_complex", filter_str, "-map", "0:v:0", "-map", "[a_out]", @@ -366,8 +460,7 @@ print(" [!] Thử lại với re-encode video...") cmd_fallback = [ self.ffmpeg, "-y", - "-i", input_video, - "-stream_loop", "-1", "-i", bgm_path, + *inputs, "-filter_complex", filter_str, "-map", "0:v:0", "-map", "[a_out]", @@ -384,15 +477,165 @@ print(f"[!] Lỗi trải BGM FFmpeg: {res2.stderr[-400:]}") return False + def mix_four_stems( + self, + video_path: str, + output_path: str, + stem1_bgm: Optional[str] = None, + stem2_ambience: Optional[str] = None, + stem3_foley: Optional[str] = None, + stem4_dialogue: Optional[str] = None, + bgm_vol: float = 0.35, + ambience_vol: float = 0.35, + foley_vol: float = 0.80, + dialogue_vol: float = 1.0, + duck_db: float = -14.0, + target_lufs: float = -14.0, + target_tp: float = -1.0, + target_lra: float = 9.0, + two_pass: bool = True + ) -> bool: + """ + Phối âm toàn diện theo Kiến Trúc 4 Stems Độc Lập chuẩn Hollywood/EBU R128: + - Stem 1 (BGM): Lặp liên tục (-stream_loop -1), EQ notch 800Hz - 3500Hz, ducking khi có thoại. + - Stem 2 (Ambience): Âm cảnh môi trường 3D liên tục, ducking khi có thoại. + - Stem 3 (Foley): Âm thanh tác động cơ học sắc nét (áo lụa, trâm cài, tách trà, vó ngựa). + - Stem 4 (Dialogue): Thoại & ngâm thơ, kích hoạt sidechain ducking (-14dB) cho Stem 1 & Stem 2. + - Master Output: Chuẩn hóa Two-Pass Linear EBU R128 (-14 LUFS, TP -1.0, LRA 9-11, 48kHz AAC). + """ + if not os.path.exists(video_path): + print(f"[!] Không tìm thấy video đầu vào: {video_path}") + return False + + os.makedirs(os.path.dirname(os.path.abspath(output_path)), exist_ok=True) + print(f"\n🎛️ BẮT ĐẦU PHỐI ÂM 4 STEMS CHO: {os.path.basename(video_path)}") + + inputs = ["-i", video_path] + filter_parts = [] + mix_inputs = [] + input_idx = 1 + + # Xác định track thoại (Stem 4 hoặc audio gốc của video) + has_dialogue_stem = stem4_dialogue is not None and os.path.exists(stem4_dialogue) + has_video_audio = self.inspect_shot_audio(video_path).get("has_audio", False) + + dialogue_source_tag = None + if has_dialogue_stem: + inputs.extend(["-i", stem4_dialogue]) + diag_idx = input_idx + input_idx += 1 + filter_parts.append(f"[{diag_idx}:a]volume={dialogue_vol}[stem4_voc]") + filter_parts.append("[stem4_voc]asplit=2[voc_main][voc_side]") + dialogue_source_tag = "[voc_side]" + mix_inputs.append("[voc_main]") + elif has_video_audio: + filter_parts.append(f"[0:a]volume={dialogue_vol}[onset_voc]") + filter_parts.append("[onset_voc]asplit=2[voc_main][voc_side]") + dialogue_source_tag = "[voc_side]" + mix_inputs.append("[voc_main]") + + # Xử lý Stem 1: BGM + if stem1_bgm and os.path.exists(stem1_bgm): + inputs.extend(["-stream_loop", "-1", "-i", stem1_bgm]) + bgm_idx = input_idx + input_idx += 1 + filter_parts.append(f"[{bgm_idx}:a]volume={bgm_vol},equalizer=f=2150:width_type=h:width=2700:g=-3.5[bgm_base]") + if dialogue_source_tag: + filter_parts.append(f"[bgm_base]{dialogue_source_tag}sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[bgm_ducked]") + mix_inputs.append("[bgm_ducked]") + else: + mix_inputs.append("[bgm_base]") + + # Xử lý Stem 2: Ambience + if stem2_ambience and os.path.exists(stem2_ambience): + inputs.extend(["-stream_loop", "-1", "-i", stem2_ambience]) + amb_idx = input_idx + input_idx += 1 + filter_parts.append(f"[{amb_idx}:a]volume={ambience_vol}[amb_base]") + if dialogue_source_tag: + filter_parts.append(f"[amb_base]{dialogue_source_tag}sidechaincompress=threshold=0.08:ratio=4:attack=50:release=300[amb_ducked]") + mix_inputs.append("[amb_ducked]") + else: + mix_inputs.append("[amb_base]") + + # Xử lý Stem 3: Foley + if stem3_foley and os.path.exists(stem3_foley): + inputs.extend(["-i", stem3_foley]) + fol_idx = input_idx + input_idx += 1 + filter_parts.append(f"[{fol_idx}:a]volume={foley_vol}[fol_base]") + mix_inputs.append("[fol_base]") + + # Nếu không có thêm stem nào, chỉ cần normalize loudness + if not mix_inputs: + return self.normalize_loudness(video_path, output_path, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra, two_pass=two_pass) + + # Trộn tất cả active stems + amix_line = f"{''.join(mix_inputs)}amix=inputs={len(mix_inputs)}:duration=first:dropout_transition=2[a_mix]" + filter_parts.append(amix_line) + filter_parts.append("[a_mix]loudnorm=I=-14:TP=-1.0:LRA=9[a_out]") + filter_complex_str = ";".join(filter_parts) + + cmd = [ + self.ffmpeg, "-y", + *inputs, + "-filter_complex", filter_complex_str, + "-map", "0:v:0", + "-map", "[a_out]", + "-c:v", "copy", + "-c:a", "aac", + "-b:a", "192k", + "-ar", "48000", + "-shortest", + output_path + ] + + res = subprocess.run(cmd, capture_output=True, text=True) + if res.returncode == 0: + print(f"[✓] Phối 4 stems thành công: {output_path}") + if two_pass: + # Chạy Pass 2 linear hoàn thiện trên master output + temp_master = output_path + ".temp_norm.mp4" + if os.path.exists(output_path): + import shutil + shutil.move(output_path, temp_master) + ok = self.normalize_loudness(temp_master, output_path, target_lufs=target_lufs, target_tp=target_tp, target_lra=target_lra, two_pass=True) + if os.path.exists(temp_master): + os.unlink(temp_master) + return ok + return True + else: + print(" [!] Thử lại với re-encode video...") + cmd_fallback = [ + self.ffmpeg, "-y", + *inputs, + "-filter_complex", filter_complex_str, + "-map", "0:v:0", + "-map", "[a_out]", + "-c:v", "libx264", "-crf", "18", "-preset", "slow", + "-c:a", "aac", "-b:a", "192k", "-ar", "48000", + "-shortest", + output_path + ] + res2 = subprocess.run(cmd_fallback, capture_output=True, text=True) + return res2.returncode == 0 if __name__ == "__main__": - parser = argparse.ArgumentParser(description="KieuStory Audio Continuity & Soundscape Engine") + parser = argparse.ArgumentParser(description="KieuStory Audio Continuity & Soundscape Engine (4-Stem & EBU R128)") parser.add_argument("--inspect", nargs="+", help="Danh sách video shot cần kiểm tra âm thanh") parser.add_argument("--stitch", nargs="+", help="Ghép nối danh sách video với crossfade âm thanh") parser.add_argument("--output", default="04_Assets/videos/master_audio_seamless.mp4", help="Đường dẫn xuất") parser.add_argument("--crossfade", type=float, default=1.0, help="Thời gian crossfade âm thanh (giây)") parser.add_argument("--l-cut", nargs=2, help="Tạo L-Cut giữa 2 video (Shot Kiều và Shot Vân)") + parser.add_argument("--normalize", type=str, help="Chuẩn hóa âm lượng EBU R128 (-14 LUFS) Two-Pass cho 1 video") + parser.add_argument("--measure", type=str, help="Đo kiểm âm lượng Pass 1 JSON qua FFmpeg loudnorm") + parser.add_argument("--mix-stems", action="store_true", help="Kích hoạt bộ trộn 4 Stems") + parser.add_argument("--video", type=str, help="Video nguồn cho mix-stems") + parser.add_argument("--bgm", type=str, help="Stem 1: BGM audio path") + parser.add_argument("--ambience", type=str, help="Stem 2: Ambience audio path") + parser.add_argument("--foley", type=str, help="Stem 3: Foley audio path") + parser.add_argument("--dialogue", type=str, help="Stem 4: Dialogue audio path") args = parser.parse_args() engine = AudioContinuityEngine() @@ -403,6 +646,19 @@ engine.stitch_with_audio_crossfade(args.stitch, args.output, crossfade_dur=args.crossfade) elif args.l_cut: engine.create_l_cut_bridge(args.l_cut[0], args.l_cut[1], args.output) + elif args.normalize: + engine.normalize_loudness(args.normalize, args.output, target_lufs=-14.0, two_pass=True) + elif args.measure: + m = engine.measure_loudness(args.measure) + print(json.dumps(m, indent=2)) + elif args.mix_stems and args.video: + engine.mix_four_stems( + args.video, args.output, + stem1_bgm=args.bgm, + stem2_ambience=args.ambience, + stem3_foley=args.foley, + stem4_dialogue=args.dialogue + ) else: # Mặc định: Kiểm tra chuỗi video Scene 1 scene1_shots = [