import glob
import os
import re
import json

inbox_dir = r"docs/design_requests/inbox"
files = sorted(glob.glob(os.path.join(inbox_dir, "*.md")))

drq_data = []

for f in files:
    content = open(f, "r", encoding="utf-8").read()
    req_id = os.path.splitext(os.path.basename(f))[0]

    title_match = re.search(r"\*\*Tên Yêu Cầu \(Title\)\*\*:\s*(.+)", content)
    title = title_match.group(1).strip() if title_match else ""

    dept_match = re.search(r"\*\*Phòng Ban Khởi Tạo \(Requesting Dept\)\*\*:\s*(.+)", content)
    dept = dept_match.group(1).strip() if dept_match else ""

    cat_match = re.search(r"\*\*Phân Loại Tài Nguyên \(Asset Category\)\*\*:\s*`?([^`\n]+)`?", content)
    cat = cat_match.group(1).strip() if cat_match else ""

    prio_match = re.search(r"\*\*Mức Độ Ưu Tiên \(Priority\)\*\*:\s*`?([^`\n]+)`?", content)
    prio = prio_match.group(1).strip() if prio_match else ""

    deliv_match = re.search(r"## 4\. DANH MỤC SẢN PHẨM ĐẦU RA MONG ĐỢI.*?\n(.*?)(?=\n## 5\.|\Z)", content, re.DOTALL)
    delivs = []
    if deliv_match:
        delivs = re.findall(r"- \[[ x]\] `?([^`\n]+)`?", deliv_match.group(1))

    specs_match = re.search(r"## 3\. THÔNG SỐ KỸ THUẬT CHI TIẾT.*?\n(.*?)(?=\n## 4\.|\Z)", content, re.DOTALL)
    specs = {}
    if specs_match:
        rows = re.findall(r"\|\s*\*\*?([^*|]+)\*\*?\s*\|\s*([^|\n]+)\s*\|", specs_match.group(1))
        specs = {k.strip(): v.strip() for k, v in rows}

    drq_data.append({
        "id": req_id,
        "title": title,
        "dept": dept,
        "cat": cat,
        "priority": prio,
        "specs": specs,
        "delivs": delivs
    })

out_json = r".agents/teamwork/spec_miner_techart_3/drqs.json"
with open(out_json, "w", encoding="utf-8") as out:
    json.dump(drq_data, out, ensure_ascii=False, indent=2)

print(f"Successfully dumped {len(drq_data)} DRQs to {out_json}")
