"""
Benchmark: Pure Python Engine vs Native C++ SIMD Engine Core.
Compares tick simulation latency, throughput, and scalability with 10,000 entities.
"""

import sys
import os
import time
import math
import unittest

sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "../../server")))

from world.spatial_grid import SpatialGrid, Entity
from world.native_engine_bridge import NativeEngineBridge


def benchmark_pure_python(entity_count: int = 5000, ticks: int = 5) -> float:
    grid = SpatialGrid(cell_size=64.0)
    entities = []
    for i in range(entity_count):
        e = Entity(entity_id=i, x=float(i % 100) * 10.0, y=float(i // 100) * 10.0)
        grid.add_entity(e)
        entities.append(e)

    start_time = time.perf_counter()
    dt = 0.03333
    for _ in range(ticks):
        for e in entities:
            # Simple displacement update
            new_x = e.x + 2.0 * dt
            new_y = e.y + 1.5 * dt
            grid.update_entity_position(e, new_x, new_y)

    total_time = (time.perf_counter() - start_time) * 1000.0
    return total_time / ticks


def benchmark_native_simd(entity_count: int = 5000, ticks: int = 5) -> float:
    bridge = NativeEngineBridge()
    if not bridge.is_loaded:
        raise RuntimeError("Native DLL not loaded")

    bridge.init(cell_size=64.0)
    for i in range(entity_count):
        idx = bridge.add_entity(
            entity_id=i,
            x=float(i % 100) * 10.0,
            y=float(i // 100) * 10.0,
            z=0.0,
            move_speed=6.0,
            collision_radius=0.5,
            hp=1000.0,
        )
        bridge.set_velocity(idx, 2.0, 1.5)

    start_time = time.perf_counter()
    dt = 0.03333
    for _ in range(ticks):
        bridge.step_tick(dt)

    total_time = (time.perf_counter() - start_time) * 1000.0
    return total_time / ticks


class TestNativeEngineBenchmark(unittest.TestCase):
    def test_native_vs_python_performance(self):
        count = 5000
        py_ms = benchmark_pure_python(entity_count=count, ticks=5)
        native_ms = benchmark_native_simd(entity_count=count, ticks=5)

        print("\n" + "=" * 60)
        print(f"BENCHMARK RESULTS: {count} ACTIVE CONCURRENT ENTITIES")
        print("=" * 60)
        print(f"Pure Python Tick Latency : {py_ms:.3f} ms / tick")
        print(f"Native C++ AVX2 Latency  : {native_ms:.3f} ms / tick")
        speedup = py_ms / max(native_ms, 0.0001)
        print(f"Native Acceleration Factor: {speedup:.1f}x FASTER")
        print("=" * 60)

        # Native must be significantly faster (at least 5x to 30x)
        self.assertLess(native_ms, py_ms)
        self.assertLess(native_ms, 5.0)  # Native must easily fit in < 5ms for 5,000 entities


if __name__ == "__main__":
    unittest.main()
