import os
import sys

if sys.platform == "win32":
    sys.stdout.reconfigure(encoding="utf-8")
    sys.stderr.reconfigure(encoding="utf-8")
import base64
import io
import json
import ssl
import time
import urllib.request

from dotenv import load_dotenv
from PIL import Image

load_dotenv(".env")
api_key = os.getenv("OPENROUTER_API_KEY")

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE


def ocr_page(img_path, page_num):
    im = Image.open(img_path)
    if im.mode != "RGB":
        im = im.convert("RGB")

    im.thumbnail((1800, 1800))
    buf = io.BytesIO()
    im.save(buf, format="JPEG", quality=88)
    img_b64 = base64.b64encode(buf.getvalue()).decode("utf-8")
    data_uri = f"data:image/jpeg;base64,{img_b64}"

    prompt = f"""Bạn là chuyên gia kiểm toán và kế toán ERP xây dựng. Hãy đọc và bóc tách TOÀN BỘ thông tin chi tiết từ hình ảnh tài liệu này (đặc biệt là Hoá đơn giá trị gia tăng, Biên bản bàn giao máy móc, Đăng ký xe/máy công trình, Giấy kiểm định, Bảng kê danh mục thiết bị, Hợp đồng kinh tế).
Lưu ý: Bóc tách chính xác các loại máy móc thiết bị như: Máy đào (bánh xích, bánh lốp), xe ô tô tải (tự đổ, ben), máy hàn (điện tử, một chiều), máy khoan (bê tông, cọc), máy cắt bê tông, máy đầm cóc/bàn, máy phát điện, máy nén khí, v.v. Kèm theo số lượng, đơn giá, thành tiền, số hoá đơn, ngày tháng năm, bên bán, bên mua, biển số, số khung số máy nếu có.

Trả về kết quả bằng định dạng JSON thuần (KHÔNG dùng markdown code block ```json) theo đúng schema:
{{
  "page_num": {page_num},
  "doc_type": "hoá đơn / đăng ký xe / kiểm định / bảng kê thiết bị / biên bản nghiệm thu / hợp đồng / khác",
  "title": "Tiêu đề tài liệu trên trang",
  "invoice_number": "số hoá đơn nếu có (ví dụ 0001234)",
  "invoice_series": "ký hiệu hoá đơn nếu có (ví dụ 1C24TDS, AA/13P)",
  "issue_date": "ngày phát hành YYYY-MM-DD nếu có",
  "seller_name": "Tên bên bán / cung cấp / chủ xe cũ",
  "seller_tax_code": "Mã số thuế bên bán nếu có",
  "buyer_name": "Tên bên mua / chủ sở hữu (CÔNG TY TNHH XÂY DỰNG ĐỊNH SƠN)",
  "buyer_tax_code": "Mã số thuế bên mua nếu có",
  "items": [
    {{
      "item_name": "Tên chi tiết thiết bị / máy móc / hàng hóa",
      "category": "excavator / truck / welder / drill / concrete_cutter / compactor / generator / compressor / pump / other",
      "unit": "chiếc / cái / bộ / ca / v.v.",
      "quantity": 1.0,
      "unit_price": 0.0,
      "total_amount": 0.0,
      "vat_rate": "10% hoặc 8% hoặc 0%",
      "vat_amount": 0.0,
      "specifications": "Thông số kỹ thuật / model / nhãn hiệu / công suất / biển số xe / số khung / số máy"
    }}
  ],
  "total_before_tax": 0.0,
  "vat_amount": 0.0,
  "total_amount": 0.0,
  "raw_notes": "Tóm tắt ngắn gọn nội dung văn bản trên trang"
}}
"""

    payload = {
        "model": "google/gemini-3.5-flash-lite",
        "messages": [
            {
                "role": "user",
                "content": [
                    {"type": "text", "text": prompt},
                    {"type": "image_url", "image_url": {"url": data_uri}},
                ],
            }
        ],
        "temperature": 0.0,
    }

    req = urllib.request.Request(
        "https://openrouter.ai/api/v1/chat/completions",
        data=json.dumps(payload).encode("utf-8"),
        headers={
            "Authorization": f"Bearer {api_key}",
            "Content-Type": "application/json",
        },
    )

    for attempt in range(3):
        try:
            with urllib.request.urlopen(req, timeout=40, context=ctx) as r:
                res = json.loads(r.read().decode("utf-8"))
                content = res["choices"][0]["message"]["content"].strip()
                content = content.removeprefix("```json")
                content = content.removeprefix("```")
                content = content.removesuffix("```")
                parsed = json.loads(content.strip())
                return parsed
        except Exception as e:
            print(f"  Attempt {attempt + 1} failed for page {page_num}: {e}")
            time.sleep(2)

    return {"page_num": page_num, "error": "Failed after 3 attempts"}


def main():
    pages_dir = "tmp/hsnl_pages"
    all_files = sorted(os.listdir(pages_dir), key=lambda x: int(x.split("_")[1]))
    results = []

    print(
        f"Starting OCR extraction for {len(all_files)} pages from HSNL ĐỊNH SƠN 2025.pdf..."
    )
    for idx, f in enumerate(all_files, 1):
        p = os.path.join(pages_dir, f)
        page_num = int(f.split("_")[1])
        print(
            f"[{idx}/{len(all_files)}] Processing Page {page_num} ({f})...", flush=True
        )
        res = ocr_page(p, page_num)
        results.append(res)
        print(
            f"  -> Type: {res.get('doc_type')} | Title: {res.get('title')} | Items: {len(res.get('items', []))}"
        )
        time.sleep(1)

    out_file = "tmp/hsnl_ocr_extracted_data.json"
    with open(out_file, "w", encoding="utf-8") as out:
        json.dump(results, out, ensure_ascii=False, indent=2)
    print(f"\nAll pages processed! Results saved to {out_file}")


if __name__ == "__main__":
    main()
