"""Orchestrate read-only GGPK extraction into assets/game_data/poe2_ggpk.db."""
from __future__ import annotations

import json
import struct
import urllib.request
from datetime import datetime, timezone
from pathlib import Path

from .bundle import decompress_bundle, parse_bundle_header
from .datc64 import load_schema, parse_datc64, poe2_table_schema
from .ggpk_fs import GgpkFs
from .hashcache import parse_hashcache
from .index_bin import decompress_index_blob
from .oodle import oodle_backend
from .paths import find_hashcache, find_oodle_dll
from .sqlite_export import (
    connect,
    init_catalog_schema,
    insert_ggpk_files,
    insert_hashcache,
    insert_logical_files,
    upsert_meta,
    write_dat_table,
    write_projections,
)

SCHEMA_URL = "https://github.com/poe-tool-dev/dat-schema/releases/download/latest/schema.min.json"

CLIENT_PATCH = "0.5.5"
CLIENT_LEAGUE = "Forbidden Rites"

DEFAULT_TABLES = (
    "BaseItemTypes",
    "ItemClasses",
    "WorldAreas",
    "Mods",
    "Stats",
    "Tags",
    "MonsterVarieties",
    "DefaultMonsterStats",
    "SkillGems",
    "ActiveSkills",
    "Quest",
    "QuestStates",
    "Words",
    "UniqueStashLayout",
    "ArmourTypes",
    "WeaponTypes",
    "CurrencyItems",
    "NPCs",
    "SoulCores",
    "SoulCoreStats",
    "SoulCoreTypes",
    "SoulCoreLimits",
    "EndgameMapTablets",
    "GoldConstants",
)


def repo_root() -> Path:
    return Path(__file__).resolve().parents[3]


def ensure_schema(schema_path: Path) -> Path:
    schema_path.parent.mkdir(parents=True, exist_ok=True)
    if schema_path.is_file() and schema_path.stat().st_size > 1000:
        return schema_path
    urllib.request.urlretrieve(SCHEMA_URL, schema_path)
    return schema_path


def find_bundle_entry(entries, bundle_name: str, fs=None):
    """Map an index bundle name (e.g. ``data``) onto a GGPK FILE record.

    PoE2 stores the dat bundle as ``Bundles2/Folders/data.dat.bundle.bin``.
    Hashed bundles may sit deeper than the catalog walk; ``fs.find`` is the fallback.
    """
    want = bundle_name.replace("\\", "/").lstrip("/")
    by_path = {e.path.replace("\\", "/"): e for e in entries}

    def _try(path: str):
        path = path.replace("\\", "/").replace("//", "/").lstrip("/")
        ent = by_path.get(path)
        if ent is not None and ent.tag == "FILE":
            return ent
        if fs is not None:
            found = fs.find(path, max_depth=6)
            if found is not None and found.tag == "FILE":
                return found
        return None

    bases = [want]
    leaf = want.split("/")[-1]
    if leaf not in bases:
        bases.append(leaf)
    prefixes = ("", "Bundles2/", "Bundles2/Folders/")
    suffixes = ("", ".bundle.bin", ".dat.bundle.bin")
    seen: set[str] = set()
    for base in bases:
        for prefix in prefixes:
            for suffix in suffixes:
                key = f"{prefix}{base}{suffix}"
                if key in seen:
                    continue
                seen.add(key)
                ent = _try(key)
                if ent is not None:
                    return ent
    leaf_l = leaf.lower()
    wanted_names = {leaf_l, f"{leaf_l}.bundle.bin", f"{leaf_l}.dat.bundle.bin"}
    for path, ent in by_path.items():
        if ent.tag != "FILE":
            continue
        if ent.name.lower() in wanted_names:
            return ent
        pl = path.replace("\\", "/").lower()
        if pl.endswith(f"/{leaf_l}.bundle.bin") or pl.endswith(f"/{leaf_l}.dat.bundle.bin"):
            return ent
    return None


def select_canonical_dat_files(logical, tables: tuple[str, ...]):
    """One .datc64 per table: prefer ``data/<name>.datc64`` over localisation copies."""
    wanted = {t.lower() for t in tables}
    best: dict[str, tuple] = {}
    for lf in logical:
        path = lf.path.replace("\\", "/").lower()
        name = path.split("/")[-1]
        if not name.endswith(".datc64"):
            continue
        stem = name[: -len(".datc64")]
        if stem not in wanted:
            continue
        exact = 0 if path == f"data/{stem}.datc64" else (1 if "/data/" in path else 2)
        score = (exact, len(path), path)
        prev = best.get(stem)
        if prev is None or score < prev[0]:
            best[stem] = (score, lf)
    return [best[t.lower()][1] for t in tables if t.lower() in best]


def extract(
    ggpk_path: Path,
    db_path: Path,
    schema_path: Path,
    tables: tuple[str, ...] = DEFAULT_TABLES,
    catalog_only: bool = False,
    log=None,
) -> dict:
    if log is None:
        log = lambda *a, **_k: print(*a, flush=True)
    stats: dict = {"ggpk": str(ggpk_path), "db": str(db_path), "status": "catalog"}
    st = ggpk_path.stat()
    conn = connect(db_path)
    try:
        init_catalog_schema(conn)
        upsert_meta(
            conn,
            {
                "extracted_at": datetime.now(timezone.utc).isoformat(),
                "ggpk_path": str(ggpk_path),
                "ggpk_size": st.st_size,
                "ggpk_mtime": int(st.st_mtime),
                "oodle_backend": "pending",
                "protocol": "ggpk-bundles2-datc64",
                "client_patch": CLIENT_PATCH,
                "client_league": CLIENT_LEAGUE,
            },
        )
        with GgpkFs(ggpk_path) as fs:
            entries = list(fs.walk(max_depth=2))
            n_files = insert_ggpk_files(conn, entries)
            log(f"[catalog] {n_files} GGPK records")
            stats["ggpk_records"] = n_files

            hc = find_hashcache(ggpk_path)
            if hc:
                recs = parse_hashcache(hc)
                insert_hashcache(conn, recs)
                log(f"[hashcache] {len(recs)} files from {hc}")
                stats["hashcache_records"] = len(recs)

            index_ent = next((e for e in entries if e.path.replace("\\", "/") == "Bundles2/_.index.bin"), None)
            data_ent = next(
                (e for e in entries if e.path.replace("\\", "/") == "Bundles2/Folders/data.dat.bundle.bin"),
                None,
            )
            if schema_path.is_file() and schema_path.stat().st_size > 1000:
                try:
                    sch = load_schema(schema_path)
                    upsert_meta(
                        conn,
                        {
                            "schema_version": sch.get("version"),
                            "schema_created_at": sch.get("createdAt"),
                            "schema_tables": len(sch.get("tables") or []),
                        },
                    )
                except (OSError, ValueError, json.JSONDecodeError):
                    pass

            if index_ent:
                hdr = parse_bundle_header(fs.read_payload_prefix(index_ent, 12 + 4096))
                upsert_meta(
                    conn,
                    {
                        "index_uncompressed": hdr.uncompressed_size,
                        "index_blocks": hdr.block_count,
                        "index_encoder": hdr.encoder,
                    },
                )
                stats["index_uncompressed"] = hdr.uncompressed_size
            if data_ent:
                hdr = parse_bundle_header(fs.read_payload_prefix(data_ent, 12 + 4096))
                upsert_meta(
                    conn,
                    {
                        "data_bundle_uncompressed": hdr.uncompressed_size,
                        "data_bundle_blocks": hdr.block_count,
                    },
                )

            conn.commit()
            if catalog_only:
                upsert_meta(conn, {"oodle_backend": "skipped-catalog-only"})
                conn.commit()
                stats["status"] = "catalog-only"
                return stats

            if find_oodle_dll() is None:
                upsert_meta(conn, {"oodle_backend": "missing"})
                conn.commit()
                stats["status"] = "catalog-no-oodle"
                stats["error"] = "Oodle DLL missing"
                return stats

            if index_ent is None:
                raise FileNotFoundError("Bundles2/_.index.bin missing inside Content.ggpk")

            log("[index] decompressing _.index.bin (Leviathan)...")
            bundles, logical = decompress_index_blob(fs.read_payload(index_ent))
            selected = select_canonical_dat_files(logical, tables)
            insert_logical_files(conn, selected)
            upsert_meta(
                conn,
                {
                    "oodle_backend": oodle_backend(),
                    "logical_files": len(logical),
                    "bundles": len(bundles),
                    "datc64_selected": len(selected),
                },
            )
            log(f"[index] {len(bundles)} bundles, {len(logical)} logical files")
            log(f"[dat] selected {len(selected)} / {len(tables)} canonical tables")
            stats["logical_files"] = len(logical)

            schema = load_schema(ensure_schema(schema_path))
            if not selected:
                sample = [
                    f.path
                    for f in logical
                    if f.path.replace("\\", "/").lower().endswith(".datc64")
                ][:20]
                log(f"[dat] no table match; datc64 sample={sample}")

            bundle_cache: dict[str, bytes] = {}
            extracted = 0
            for lf in selected:
                table_name = Path(lf.path).stem  # BaseItemTypes
                # datc64 stem keeps original casing from path; normalize via requested list
                canon = next((t for t in tables if t.lower() == table_name.lower()), table_name)
                spec = poe2_table_schema(schema, canon)
                if spec is None:
                    log(f"[dat] skip {lf.path}: no schema")
                    continue
                entry = find_bundle_entry(entries, lf.bundle_name, fs)
                if entry is None:
                    log(f"[dat] skip {lf.path}: bundle {lf.bundle_name!r} not in GGPK")
                    continue
                raw_bundle = bundle_cache.get(entry.path)
                if raw_bundle is None:
                    log(f"[bundle] decompress {entry.path} ({entry.data_size:,} B)...")
                    raw_bundle = decompress_bundle(fs.read_payload(entry))
                    bundle_cache[entry.path] = raw_bundle
                slice_bytes = raw_bundle[lf.offset : lf.offset + lf.size]
                try:
                    rows = parse_datc64(slice_bytes, spec)
                except (ValueError, OverflowError, struct.error) as exc:
                    log(f"[dat] skip {canon}: {exc}")
                    continue
                write_dat_table(conn, canon, rows)
                conn.commit()
                log(f"[dat] {canon}: {len(rows)} rows")
                extracted += 1
            write_projections(conn)
            upsert_meta(conn, {"dat_tables": extracted, "status": "complete"})
            conn.commit()
            stats["dat_tables"] = extracted
            stats["status"] = "complete"
            return stats
    finally:
        conn.close()
