import os
import sys
import json
import time
import torch
import whisper

if hasattr(sys.stdout, 'reconfigure'):
    sys.stdout.reconfigure(encoding='utf-8')

venv_scripts = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'venv', 'Scripts')
if venv_scripts not in os.environ.get('PATH', ''):
    os.environ['PATH'] = venv_scripts + os.pathsep + os.environ.get('PATH', '')

output_dir = os.path.join('content', 'subtitles', 'raw_json')
os.makedirs(output_dir, exist_ok=True)

device = 'cuda' if torch.cuda.is_available() else 'cpu'
print(f"Loading Whisper 'base' on {device}...")
model = whisper.load_model('base', device=device)

PROMPT = (
    "Trận thủy chiến Rạch Gầm - Xoài Mút năm 1785. Nguyễn Huệ, Nguyễn Nhạc, Nguyễn Ánh, "
    "Chiêu Tăng, Chiêu Sương, tướng Trương Văn Đa, quân Tây Sơn, quân Xiêm, Gia Định, "
    "Quy Nhơn, Mỹ Tho, Chân Lạp, hỏa pháo, thuyền chiến, của ta, sông Tiền, Cù lao Thới Sơn."
)

for ep in range(1, 9):
    video_path = os.path.join('content', 'videos', f'ep{ep}.mp4')
    if not os.path.exists(video_path):
        print(f"[-] Video not found: {video_path}")
        continue
    
    out_json = os.path.join(output_dir, f'ep{ep}_raw.json')
    print(f"\n[{ep}/8] Transcribing {video_path}...")
    t0 = time.time()
    
    res = model.transcribe(
        video_path,
        language='vi',
        initial_prompt=PROMPT,
        condition_on_previous_text=False,
        no_speech_threshold=0.6,
        compression_ratio_threshold=2.4
    )
    t1 = time.time()
    print(f"[{ep}/8] Finished in {t1-t0:.2f}s! Found {len(res['segments'])} segments.")
    
    data = {
        "episode": ep,
        "video": video_path,
        "duration": res.get("duration", 0),
        "segments": [
            {
                "id": s["id"],
                "start": round(s["start"], 2),
                "end": round(s["end"], 2),
                "text": s["text"].strip(),
                "no_speech_prob": round(s.get("no_speech_prob", 0), 3)
            }
            for s in res["segments"]
        ]
    }
    
    with open(out_json, 'w', encoding='utf-8') as f:
        json.dump(data, f, ensure_ascii=False, indent=2)
    print(f"Saved: {out_json}")

print("\n✓ ALL EPISODES TRANSCRIBED SUCCESSFULLY!")
