import os
import sys

def main():
    sys.stdout.reconfigure(encoding='utf-8')

    sources = [
        {
            'file_name': '01_chinh_sach_kiem_tien_kenh_youtube.md',
            'title': 'Chính sách về việc kiếm tiền trên kênh YouTube',
            'url': 'https://support.google.com/youtube/answer/1311392?hl=vi',
            'path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\8\output.txt',
            'start_marker': '# Chính sách về việc kiếm tiền trên kênh YouTube',
            'end_marker': '## Thông tin này có hữu ích không?'
        },
        {
            'file_name': '02_nguyen_tac_noi_dung_phu_hop_nha_quang_cao.md',
            'title': 'Nguyên tắc về nội dung phù hợp với nhà quảng cáo (Advertiser-Friendly Guidelines)',
            'url': 'https://support.google.com/youtube/answer/6162278?hl=vi',
            'path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\16\output.txt',
            'start_marker': '# Nguyên tắc về nội dung phù hợp với nhà quảng cáo',
            'end_marker': '## Thông tin này có hữu ích không?'
        },
        {
            'file_name': '03_dieu_kien_tham_gia_chuong_trinh_doi_tac_ypp.md',
            'title': 'Tổng quan và điều kiện tham gia Chương trình Đối tác YouTube (YPP)',
            'url': 'https://support.google.com/youtube/answer/72851?hl=vi',
            'path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\18\output.txt',
            'start_marker': '# Tổng quan và điều kiện tham gia Chương trình Đối tác YouTube',
            'end_marker': '## Thông tin này có hữu ích không?'
        },
        {
            'file_name': '04_chinh_sach_kiem_tien_youtube_shorts.md',
            'title': 'Chính sách về việc kiếm tiền trên YouTube Shorts',
            'url': 'https://support.google.com/youtube/answer/12504220?hl=vi',
            'path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\20\output.txt',
            'start_marker': '# Chính sách về việc kiếm tiền trên YouTube Shorts',
            'end_marker': '## Thông tin này có hữu ích không?'
        },
        {
            'file_name': '05_cong_bo_su_dung_ai_tao_sinh.md',
            'title': 'Công bố việc sử dụng nội dung dựa trên AI tạo sinh',
            'url': 'https://support.google.com/youtube/answer/14328491?hl=vi',
            'path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\26\output.txt',
            'start_marker': '# Công bố việc sử dụng nội dung dựa trên AI tạo sinh',
            'end_marker': '## Thông tin này có hữu ích không?'
        },
        {
            'file_name': '06_chinh_sach_chuong_trinh_adsense.md',
            'title': 'Chính sách chương trình của AdSense',
            'url': 'https://support.google.com/adsense/answer/48182?hl=vi',
            'path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\44\output.txt',
            'start_marker': '# Chính sách chương trình của AdSense',
            'end_marker': '## Thông tin này có hữu ích không?'
        },
        {
            'file_name': '07_nguyen_tac_cong_dong_va_ban_quyen.md',
            'title': 'Nguyên tắc cộng đồng và Bản quyền trên YouTube',
            'url': 'https://support.google.com/youtube/answer/9288567?hl=vi',
            'path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\22\output.txt',
            'start_marker': '# Nguyên tắc cộng đồng của YouTube',
            'end_marker': '## Thông tin này có hữu ích không?',
            'extra_path': r'C:\Users\Admin\.gemini\antigravity\brain\c40ffad9-9d27-4699-98d1-215989e724f2\.system_generated\steps\24\output.txt',
            'extra_start': '# Bản quyền trên YouTube',
            'extra_end': '## Thông tin này có hữu ích không?'
        }
    ]

    out_dir = r'c:\Projects\Historical\yt-rule'
    os.makedirs(out_dir, exist_ok=True)

    for item in sources:
        with open(item['path'], 'r', encoding='utf-8') as f:
            content = f.read()

        start_idx = content.find(item['start_marker'])
        if start_idx == -1:
            print(f"Error finding start marker for {item['file_name']}")
            continue

        end_idx = content.find(item['end_marker'], start_idx)
        body = content[start_idx:end_idx] if end_idx != -1 else content[start_idx:]

        extra_body = ""
        if 'extra_path' in item:
            with open(item['extra_path'], 'r', encoding='utf-8') as f:
                econtent = f.read()
            estart = econtent.find(item['extra_start'])
            if estart != -1:
                eend = econtent.find(item['extra_end'], estart)
                extra_body = "\n\n---\n\n" + (econtent[estart:eend] if eend != -1 else econtent[estart:])

        # Clean document
        header = f"""# {item['title']}

> **Nguồn trích xuất chính thức từ YouTube / Google Support**: [{item['url']}]({item['url']})  
> **Thời điểm cập nhật & lưu trữ**: 2026-10-09  
> **Phạm vi áp dụng**: Quy chuẩn kiếm tiền chính thức dành cho nhà sáng tạo nội dung YouTube.

---

"""
        # Remove redundant H1 if the body already starts with H1
        lines = body.strip().splitlines()
        if lines and lines[0].startswith('# '):
            body_cleaned = '\n'.join(lines[1:]).strip()
        else:
            body_cleaned = body.strip()

        full_text = header + body_cleaned + extra_body.rstrip() + '\n'

        target_path = os.path.join(out_dir, item['file_name'])
        with open(target_path, 'w', encoding='utf-8') as f:
            f.write(full_text)
        print(f"Wrote {item['file_name']} ({len(full_text.splitlines())} lines)")

if __name__ == '__main__':
    main()
