"""GD1 - Phan tich tinh cac binary thay doi/moi cua JxVoCongTruyenKy so voi baseline 2026-09-18.

READ-ONLY. Xuat bao cao vao security-baseline/2026-09-28-vctk/static_analysis/.
Nguon doi chieu "ban cu": Full_JxVCTK_123456.zip (doc truc tiep trong zip, khong giai nen).
"""
from __future__ import annotations

import collections
import hashlib
import json
import math
import re
import time
import zipfile
from pathlib import Path

GAME = Path(r"C:\JXVCTK\JxVoCongTruyenKy")
ZIP = Path(r"C:\JXVCTK\Full_JxVCTK_123456.zip")
OUT = Path(r"C:\Projects\JX\security-baseline\2026-09-28-vctk\static_analysis")
OUT.mkdir(parents=True, exist_ok=True)

STR_RE = re.compile(rb"[\x20-\x7e]{5,}")
IP_RE = re.compile(rb"\b(?:(?:25[0-5]|2[0-4]\d|[01]?\d\d?)\.){3}(?:25[0-5]|2[0-4]\d|[01]?\d\d?)\b")
URL_RE = re.compile(rb"https?://[\x21-\x7e]{4,200}")

KEYWORDS = [
    "http://", "https://", "webhook", "telegram", "discord", "wallet.dat", "login data",
    "cookies", "cryptunprotect", "credenumerate", "minidump", "writeprocessmemory",
    "readprocessmemory", "createremotethread", "virtualprotectex", "setwindowshookex",
    "getasynckeystate", "keybd_event", "mouse_event", "sendinput", "shellexecute",
    "createprocess", "winhttp", "internetopen", "urldownload", "loadlibrary",
    "getprocaddress", "createservice", "schtasks", "powershell", "cmd.exe",
    "appdata", "startup", "currentversion\\run", "vctk.me", "vietguards", "kimyen",
    "congdongvolam", "vlbs", "trainjx", "vggame", "game64", "http", 
]

# (nhan, duong dan hien tai tren dia)
TARGETS = [
    ("launcher_new_AutoUpdateVCTK.exe", GAME / "AutoUpdateVCTK.exe"),
    ("vlhookpr_new_VLHookPr.dll", GAME / "_VLAuto" / "VLHookPr.dll"),
    ("vlhookpr_backup_VLHookPr.dll.backup", GAME / "_VLAuto" / "VLHookPr.dll.backup"),
    ("vlautopr_new_VLAutoPr.exe", GAME / "_VLAuto" / "VLAutoPr.exe"),
    ("kymyen_new_KY_VCTK_CTCX_NEW.exe", GAME / "_AutoKimYen" / "KY_VCTK_CTCX_NEW.exe"),
    ("kymyen_new_vauth.auto", GAME / "_AutoKimYen" / "vauth.auto"),
    ("asi_mPK_congdongvolam.com.asi", GAME / "_AutoVLBS" / "VLBS19Dai" / "mPK_congdongvolam.com.asi"),
    ("vlbs13_core.dll", GAME / "_AutoVLBS" / "VLBS13" / "core.dll"),
    ("winmm_VLBS19Dai.dll", GAME / "_AutoVLBS" / "VLBS19Dai" / "winmm.dll"),
    ("blob_vietguards.driver", GAME / "vietguards.driver"),
    ("vietguard_VietGuardJX.sys", GAME / "VietGuardJX.sys"),
    ("engine_engine.dll", GAME / "engine.dll"),
    ("film_Game_film.exe", GAME / "Game_film.exe"),
    ("rainbow_Rainbow.dll", GAME / "Rainbow.dll"),
    ("vggame_current_vggame.exe", GAME / "vggame.exe"),
]

# File cu trong zip de doi chieu (zip root 'JxVoCongTruyenKy/')
ZIP_OLD = [
    ("old_launcher_AutoUpdateVCTK.exe", "JxVoCongTruyenKy/AutoUpdateVCTK.exe"),
    ("old_vlautopr_VLAutoPr.exe", "JxVoCongTruyenKy/_VLAuto/VLAutoPr.exe"),
    ("old_vlhookpr_VLHookPr.dll", "JxVoCongTruyenKy/_VLAuto/VLHookPr.dll"),
    ("old_kymyen_KY_VCTK_CTCX_NEW.exe", "JxVoCongTruyenKy/_AutoKimYen/KY_VCTK_CTCX_NEW.exe"),
    ("old_winmm_winmm.dll", "JxVoCongTruyenKy/_AutoVLBS/VLBS19Dai/winmm.dll"),
    ("old_driver_vietguards.driver", "JxVoCongTruyenKy/vietguards.driver"),
    ("old_vggame_vggame.exe", "JxVoCongTruyenKy/vggame.exe"),
    ("old_engine_engine.dll", "JxVoCongTruyenKy/engine.dll"),
]

# Zip nguon la goi phat hanh co mat khau; thu lan luot cac ung vien (ten file chua '123456')
ZIP_PWD_CANDIDATES = [None, b"123456"]


def sha256_md5(buf: bytes):
    return hashlib.sha256(buf).hexdigest(), hashlib.md5(buf).hexdigest()


def entropy(buf: bytes) -> float:
    if not buf:
        return 0.0
    c = collections.Counter(buf)
    n = len(buf)
    return -sum((v / n) * math.log2(v / n) for v in c.values())


def carve_strings(buf: bytes):
    return [m.group().decode("latin1") for m in STR_RE.finditer(buf)]


def interesting(strings, limit_per_bucket=25):
    urls = sorted({s for s in strings if s.lower().startswith(("http://", "https://", "ftp://"))})
    ips = sorted({s for s in strings if IP_RE.fullmatch(s.encode("latin1", "ignore")) or IP_RE.search(s.encode("latin1", "ignore"))})
    # IPs: only pure-IP-like strings
    ips = sorted({s for s in strings if re.fullmatch(r"(?:(?:25[0-5]|2[0-4]\d|[01]?\d\d?)\.){3}(?:25[0-5]|2[0-4]\d|[01]?\d\d?)", s)})
    hits = {}
    low = [s.lower() for s in strings]
    for kw in KEYWORDS:
        n = sum(1 for s in low if kw in s)
        if n:
            hits[kw] = n
    return {
        "urls": urls[:limit_per_bucket],
        "url_count": len(urls),
        "ips": ips[:limit_per_bucket],
        "ip_count": len(ips),
        "keyword_hits": dict(sorted(hits.items(), key=lambda kv: -kv[1])),
    }

def pe_report(buf: bytes):
    rep = {"is_pe": False}
    try:
        import pefile

        pe = pefile.PE(data=buf, fast_load=True)
        rep["is_pe"] = True
        rep["machine"] = hex(pe.FILE_HEADER.Machine)
        rep["timestamp"] = pe.FILE_HEADER.TimeDateStamp
        rep["subsystem"] = pe.OPTIONAL_HEADER.Subsystem
        rep["ep_rva"] = hex(pe.OPTIONAL_HEADER.AddressOfEntryPoint)
        rep["image_base"] = hex(pe.OPTIONAL_HEADER.ImageBase)
        secs = []
        ents = []
        for s in pe.sections:
            nm = s.Name.split(b"\x00", 1)[0].decode("latin1", "replace")
            data = s.get_data()
            ent = entropy(data[: 1 << 20])
            ents.append(ent)
            ch = int(s.Characteristics)
            rwx = bool(ch & 0x20000000) and bool(ch & 0x80000000) and bool(ch & 0x40000000)
            secs.append(
                "%s vsz=0x%X raw=0x%X ent=%.2f%s"
                % (nm if nm else "(empty)", s.Misc_VirtualSize, s.SizeOfRawData, ent, " RWX!" if rwx else "")
            )
        rep["sections"] = secs
        rep["max_section_entropy"] = round(max(ents) if ents else 0.0, 2)
        try:
            import pefile as pf

            pe.parse_data_directories(
                directories=[
                    pf.DIRECTORY_ENTRY["IMAGE_DIRECTORY_ENTRY_IMPORT"],
                    pf.DIRECTORY_ENTRY["IMAGE_DIRECTORY_ENTRY_EXPORT"],
                    pf.DIRECTORY_ENTRY["IMAGE_DIRECTORY_ENTRY_SECURITY"],
                ]
            )
            imps = {}
            if hasattr(pe, "DIRECTORY_ENTRY_IMPORT"):
                for e in pe.DIRECTORY_ENTRY_IMPORT:
                    dll = e.dll.decode("latin1", "replace") if e.dll else "?"
                    imps[dll] = [
                        (i.name.decode("latin1", "replace") if i.name else "ord_%s" % i.ordinal)
                        for i in e.imports
                    ]
            rep["imports"] = imps
            rep["import_dlls"] = sorted(imps)
            if hasattr(pe, "DIRECTORY_ENTRY_EXPORT"):
                rep["export_count"] = len(pe.DIRECTORY_ENTRY_EXPORT.symbols)
            sec_dir = pe.OPTIONAL_HEADER.DATA_DIRECTORY[pf.DIRECTORY_ENTRY["IMAGE_DIRECTORY_ENTRY_SECURITY"]]
            rep["has_authenticode"] = sec_dir.VirtualAddress > 0 and sec_dir.Size > 0
        except Exception as ex:
            rep["dirs_error"] = str(ex)
        try:
            ov = pe.get_overlay_data_start_offset()
            rep["overlay_size"] = (len(buf) - ov) if ov else 0
        except Exception:
            rep["overlay_size"] = None
        pe.close()
    except Exception as ex:
        rep["pe_error"] = str(ex)
    return rep

def analyze_blob(buf: bytes):
    """Phan tich blob khong phai PE: entropy theo MB, mang PE an, header hex."""
    per_mb = []
    for i in range(0, len(buf), 1 << 20):
        per_mb.append(round(entropy(buf[i : i + (1 << 20)]), 3))
    embedded = []
    start = 0
    while True:
        idx = buf.find(b"MZ", start)
        if idx < 0:
            break
        if idx + 0x40 < len(buf):
            e_lfanew = int.from_bytes(buf[idx + 0x3C : idx + 0x40], "little")
            if 0 < e_lfanew < 0x1000 and idx + e_lfanew + 4 <= len(buf):
                if buf[idx + e_lfanew : idx + e_lfanew + 4] == b"PE\x00\x00":
                    embedded.append(idx)
        start = idx + 2
        if len(embedded) > 20:
            break
    return {
        "per_mb_entropy": per_mb,
        "embedded_pe_offsets": embedded,
        "first_bytes_hex": buf[:32].hex(" "),
    }


def write_strings_file(name: str, strings):
    p = OUT / ("%s.strings.txt" % name)
    with p.open("w", encoding="utf-8", errors="replace") as f:
        for i, s in enumerate(strings):
            if i >= 40000:
                f.write("... (truncated at 40000 lines)\n")
                break
            f.write(s + "\n")
    return p.name


def process_current():
    results = {}
    for name, path in TARGETS:
        if not path.exists():
            results[name] = {"path": str(path), "error": "MISSING"}
            print("MISSING", path)
            continue
        buf = path.read_bytes()
        sha, md5 = sha256_md5(buf)
        rec = {
            "path": str(path),
            "size": len(buf),
            "mtime_local": time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(path.stat().st_mtime)),
            "sha256": sha.upper(),
            "md5": md5.upper(),
        }
        is_mz = buf[:2] == b"MZ"
        rec["mz_header"] = is_mz
        if is_mz:
            rec.update(pe_report(buf))
        else:
            rec["blob"] = analyze_blob(buf)
        strings = carve_strings(buf)
        rec["string_count"] = len(strings)
        rec["interesting"] = interesting(strings)
        rec["strings_file"] = write_strings_file(name, strings)
        results[name] = rec
        print("OK", name, rec["size"], rec["sha256"][:16])
    return results


def process_zip_old():
    results = {}
    z = zipfile.ZipFile(ZIP)
    try:
        for name, entry in ZIP_OLD:
            buf = None
            err = None
            for pwd in ZIP_PWD_CANDIDATES:
                try:
                    buf = z.read(entry, pwd=pwd)
                    break
                except KeyError:
                    err = "NOT_IN_ZIP"
                    break
                except Exception as ex:
                    err = "ZIP_READ_FAIL: %s" % ex
            if buf is None:
                results[name] = {"zip_entry": entry, "error": err}
                print("ZIP SKIP", name, err)
                continue
            sha, md5 = sha256_md5(buf)
            rec = {"zip_entry": entry, "size": len(buf), "sha256": sha.upper(), "md5": md5.upper()}
            if buf[:2] == b"MZ":
                rec.update(pe_report(buf))
            strings = carve_strings(buf)
            rec["string_count"] = len(strings)
            rec["interesting"] = interesting(strings)
            rec["strings_file"] = write_strings_file(name, strings)
            results[name] = rec
            print("ZIP OK", name, rec["size"], rec["sha256"][:16])
    finally:
        z.close()
    return results

def diff_sets(rec_new, rec_old, label):
    out = ["## Diff cu/moi: %s" % label]
    a = set(rec_new.get("import_dlls") or [])
    b = set(rec_old.get("import_dlls") or [])
    out.append("- import DLL them moi: %s" % (sorted(a - b) or "khong"))
    out.append("- import DLL mat di : %s" % (sorted(b - a) or "khong"))
    ia = set((rec_new.get("interesting") or {}).get("urls") or [])
    ib = set((rec_old.get("interesting") or {}).get("urls") or [])
    out.append("- URL moi xuat hien: %s" % (sorted(ia - ib)[:20] or "khong"))
    out.append("- URL cu mat di    : %s" % (sorted(ib - ia)[:20] or "khong"))
    ims_new = rec_new.get("imports") or {}
    ims_old = rec_old.get("imports") or {}
    for dll in sorted(set(ims_new) & set(ims_old)):
        fa = set(ims_new[dll])
        fb = set(ims_old[dll])
        add = sorted(fa - fb)
        rem = sorted(fb - fa)
        if add or rem:
            out.append("  - %s: +%s | -%s" % (dll, add[:15] or "khong", rem[:15] or "khong"))
    return out

def main():
    print("=== Current files ===")
    cur = process_current()
    print("=== Zip old files ===")
    old = process_zip_old()

    lines = []
    lines.append("# Static analysis - binary thay doi/moi (2026-09-28)")
    lines.append("")
    lines.append("Sinh luc: %s" % time.strftime("%Y-%m-%d %H:%M:%S"))
    lines.append("Nguon: `%s` + doi chieu ban cu tu `%s`" % (GAME, ZIP))
    lines.append("")
    lines.append("## Bang tong hop")
    lines.append("")
    lines.append("| Target | Size | SHA256 | PE? | MaxSecEnt | RWX | Authenticode | Ghi chu |")
    lines.append("|---|---|---|---|---|---|---|---|")
    for name, rec in cur.items():
        if rec.get("error"):
            lines.append("| %s | - | - | - | - | - | - | %s |" % (name, rec["error"]))
            continue
        note = ""
        if not rec.get("is_pe") and rec.get("mz_header"):
            note = "MZ nhung khong parse duoc PE"
        if rec.get("blob"):
            note = "KHONG phai PE (blob)"
        lines.append(
            "| %s | %s | `%s` | %s | %s | %s | %s | %s |"
            % (
                name,
                rec["size"],
                rec["sha256"][:16] + "...",
                rec.get("is_pe"),
                rec.get("max_section_entropy", "-"),
                "CO" if any("RWX!" in s for s in (rec.get("sections") or [])) else "-",
                rec.get("has_authenticode", "-"),
                note,
            )
        )
    lines.append("")
    for name, rec in cur.items():
        lines.append("## %s" % name)
        lines.append("")
        if rec.get("error"):
            lines.append("- ERROR: %s" % rec["error"])
            lines.append("")
            continue
        lines.append("- Path: `%s`" % rec["path"])
        lines.append("- Size: %s | mtime: %s" % (rec["size"], rec["mtime_local"]))
        lines.append("- SHA256: `%s`" % rec["sha256"])
        lines.append("- MD5: `%s`" % rec["md5"])
        if rec.get("is_pe"):
            lines.append(
                "- PE: machine=%s subsystem=%s ep=%s imagebase=%s authenticode=%s export=%s"
                % (rec.get("machine"), rec.get("subsystem"), rec.get("ep_rva"), rec.get("image_base"),
                   rec.get("has_authenticode"), rec.get("export_count", 0))
            )
            lines.append("- Sections: %s" % " ; ".join(rec.get("sections") or []))
            lines.append("- Import DLLs: %s" % (", ".join(rec.get("import_dlls") or []) or "-"))
            imp_lines = []
            for dll, fns in (rec.get("imports") or {}).items():
                imp_lines.append("%s: %s" % (dll, ", ".join(fns[:30])))
            lines.append("- Imports: %s" % (" || ".join(imp_lines)[:3000] or "-"))
        if rec.get("blob"):
            b = rec["blob"]
            lines.append("- BLOB per-MB entropy: %s" % b["per_mb_entropy"][:40])
            lines.append("- BLOB embedded PE offsets: %s" % (b["embedded_pe_offsets"] or "khong co"))
            lines.append("- BLOB first bytes: `%s`" % b["first_bytes_hex"])
        it = rec.get("interesting") or {}
        lines.append("- URL (%s): %s" % (it.get("url_count", 0), it.get("urls") or "khong"))
        lines.append("- IP (%s): %s" % (it.get("ip_count", 0), it.get("ips") or "khong"))
        lines.append("- Keyword hits: %s" % (it.get("keyword_hits") or "khong"))
        lines.append("- Full strings: `%s`" % rec.get("strings_file"))
        lines.append("")

    lines.append("# Diff ban cu (trong zip) vs hien tai")
    lines.append("")
    pairs = [
        ("old_launcher_AutoUpdateVCTK.exe", "launcher_new_AutoUpdateVCTK.exe", "AutoUpdateVCTK.exe"),
        ("old_vlhookpr_VLHookPr.dll", "vlhookpr_new_VLHookPr.dll", "VLHookPr.dll"),
        ("old_vlautopr_VLAutoPr.exe", "vlautopr_new_VLAutoPr.exe", "VLAutoPr.exe"),
        ("old_kymyen_KY_VCTK_CTCX_NEW.exe", "kymyen_new_KY_VCTK_CTCX_NEW.exe", "KY_VCTK_CTCX_NEW.exe"),
        ("old_winmm_winmm.dll", "winmm_VLBS19Dai.dll", "winmm.dll"),
        ("old_driver_vietguards.driver", "blob_vietguards.driver", "vietguards.driver"),
        ("old_vggame_vggame.exe", "vggame_current_vggame.exe", "vggame.exe"),
        ("old_engine_engine.dll", "engine_engine.dll", "engine.dll"),
    ]
    for oldn, newn, label in pairs:
        rnew = cur.get(newn)
        rold = old.get(oldn)
        if rnew and rold and not rnew.get("error") and not rold.get("error"):
            sha_same = rnew["sha256"] == rold["sha256"]
            lines.append("- %s: SHA256 %s (%s vs zip %s)" % (label, "GIONG NHAU" if sha_same else "KHAC NHAU", rnew["sha256"][:16], rold["sha256"][:16]))
            if not sha_same:
                lines.extend(diff_sets(rnew, rold, label))
        else:
            lines.append("- %s: thieu du lieu (zip/new)" % label)
    lines.append("")
    with (OUT / "report.md").open("w", encoding="utf-8") as f:
        f.write("\n".join(lines))
    with (OUT / "summary.json").open("w", encoding="utf-8") as f:
        json.dump({"current": cur, "zip_old": old}, f, ensure_ascii=False, indent=1)
    print("DONE ->", OUT / "report.md")


if __name__ == "__main__":
    main()






