import time
import torch
import tensorrt as trt

ENGINE_PATH = "/workspace/purescale_4k.engine"
print("Loading TensorRT Engine...")
t0 = time.time()
runtime = trt.Runtime(trt.Logger(trt.Logger.WARNING))
with open(ENGINE_PATH, "rb") as f:
    engine = runtime.deserialize_cuda_engine(f.read())
context = engine.create_execution_context()
print(f"✓ Loaded Engine in {time.time() - t0:.2f}s")

# Allocate buffers
d_input = torch.randn(1, 3, 720, 1280, device="cuda", dtype=torch.float32).contiguous()
d_output = torch.empty(1, 3, 2160, 3840, device="cuda", dtype=torch.float32).contiguous()

context.set_tensor_address("input", d_input.data_ptr())
context.set_tensor_address("output", d_output.data_ptr())

stream = torch.cuda.Stream()

# Warmup 3 frames
print("Warming up...")
for _ in range(3):
    with torch.cuda.stream(stream):
        context.execute_async_v3(stream.cuda_stream)
    stream.synchronize()

print("Benchmarking 10 frames...")
torch.cuda.synchronize()
t_start = time.time()
N = 10
for _ in range(N):
    with torch.cuda.stream(stream):
        context.execute_async_v3(stream.cuda_stream)
    stream.synchronize()

torch.cuda.synchronize()
total_time = time.time() - t_start
ms_per_frame = (total_time / N) * 1000
fps = N / total_time
print(f"✓ Output shape: {d_output.shape}, min: {d_output.min().item():.3f}, max: {d_output.max().item():.3f}")
print(f"🚀 Speed: {ms_per_frame:.1f} ms/frame ({fps:.2f} FPS) on RTX 4090!")
