#!/usr/bin/env python3
# -*- coding: utf-8 -*-
import sys
import os
import json
import re
from pathlib import Path

if sys.platform == "win32":
    try:
        sys.stdout.reconfigure(encoding="utf-8")
        sys.stderr.reconfigure(encoding="utf-8")
    except Exception:
        pass

BASE_DIR = Path(r"c:\Projects\KieuStory")

# 1. Inspect prompt files
prompt_files = [
    BASE_DIR / "episodes" / "ep01" / "prompts" / "muse_prompts.json",
    BASE_DIR / "02_AI_Prompts" / "muse_ai_video_prompts.json"
]

print("=== 1. PROMPT FILES AUDIO GUARD AUDIT ===")
CANONICAL_RULE = "Quy tắc âm thanh: Tuyệt đối KHÔNG sinh nhạc nền (no music/BGM), không âm thanh điện tử, không tạp âm rè nhiễu. Chỉ sinh âm thanh môi trường tự nhiên (foley, ambience) và thoại nhân vật chân thực."

for pf in prompt_files:
    if not pf.exists():
        continue
    data = json.loads(pf.read_text(encoding="utf-8"))
    mp_dict = data.get("motion_prompts", {})
    total = len(mp_dict)
    has_canonical = 0
    has_any_no_bgm = 0
    missing_guard = []
    
    for sid, sdata in mp_dict.items():
        if isinstance(sdata, dict):
            mp = sdata.get("motion_prompt", "")
        else:
            mp = str(sdata)
        
        if CANONICAL_RULE in mp:
            has_canonical += 1
        if "Tuyệt đối KHÔNG sinh nhạc nền" in mp or "Strictly NO background music" in mp:
            has_any_no_bgm += 1
        else:
            missing_guard.append(sid)
            
    print(f"File: {pf.relative_to(BASE_DIR)}")
    print(f"  Total prompts: {total}")
    print(f"  Has exact canonical rule: {has_canonical}/{total}")
    print(f"  Has any no-BGM guard:     {has_any_no_bgm}/{total}")
    print(f"  Missing guard count:       {len(missing_guard)}")
    if missing_guard:
        print(f"  Sample missing: {missing_guard[:5]}")

# 2. Inspect code files
print("\n=== 2. CODE IMPLEMENTATION AUDIT ===")
pipeline_files = [
    BASE_DIR / "05_Production_Pipeline" / "run_shot.py",
    BASE_DIR / "05_Production_Pipeline" / "production_orchestrator.py",
    BASE_DIR / "05_Production_Pipeline" / "multi_worker_orchestrator.py"
]

for pf in pipeline_files:
    print(f"\nFile: {pf.name}")
    content = pf.read_text(encoding="utf-8")
    
    # Check if Audio Guard is referenced or modified
    guard_matches = re.findall(r"(audio_guard|nhạc nền|no music)", content, re.IGNORECASE)
    print(f"  Occurrences of audio guard keywords: {len(guard_matches)}")
    
    # Check exact lines
    for idx, line in enumerate(content.splitlines(), 1):
        if any(kw in line.lower() for kw in ["audio_guard", "tuyệt đối không sinh nhạc nền", "quy tắc âm thanh"]):
            print(f"    Line {idx}: {line.strip()[:100]}")

