import os
import sys
import torch
import whisper

if hasattr(sys.stdout, 'reconfigure'):
    sys.stdout.reconfigure(encoding='utf-8')

device = 'cuda' if torch.cuda.is_available() else 'cpu'
print(f"Loading Whisper 'base' on {device}...")
model = whisper.load_model('base', device=device)

video_path = sys.argv[1] if len(sys.argv) > 1 else r"content\videos\ep1.mp4"
print(f"Transcribing {video_path}...")
res = model.transcribe(video_path, language='vi', word_timestamps=True)

for i, seg in enumerate(res['segments']):
    print(f"\n[{i+1}] {seg['start']:.2f}s -> {seg['end']:.2f}s: {seg['text']}")
    words = [f"{w['word']}({w['start']:.2f}-{w['end']:.2f})" for w in seg.get('words', [])]
    print("   Words: " + " ".join(words[:8]))
