# -*- coding: utf-8 -*-
import glob
import re
import sys

sys.stdout.reconfigure(encoding="utf-8")

def is_dialogue_or_quote(text: str) -> bool:
    t = text.strip()
    if not t:
        return True
    if t.startswith(("#", ">", "---")):
        return True
    if t.startswith(("—", "-", "–", '"', "“", "「", "*—", "*-", "*–")):
        return True
    # If the whole sentence or paragraph is enclosed in quotes or asterisks (thought/dialogue)
    if (t.startswith('"') and t.endswith('"')) or (t.startswith('“') and t.endswith('”')) or (t.startswith('*') and t.endswith('*')):
        return True
    return False

def scan_language(lang: str, regex_pattern: str):
    print(f"=== SCANNING {lang.upper()} ===")
    count = 0
    for fpath in sorted(glob.glob(f"novels/locales/{lang}/act_*/*.md")):
        lines = open(fpath, encoding="utf-8").readlines()
        for idx, line in enumerate(lines):
            s = line.strip()
            if is_dialogue_or_quote(s):
                continue
            # If line contains dialogue marker inside (e.g. He said: "I am ...")
            # We only care about narration part outside quotes
            # Strip out text inside quotes: "...", “...”, *— ...*
            narrative = re.sub(r'["“「].*?["”」]', '', s)
            narrative = re.sub(r'\*—.*?\*', '', narrative)
            matches = list(re.finditer(regex_pattern, narrative, re.IGNORECASE))
            if matches:
                # filter out false positives if needed
                filtered = []
                for m in matches:
                    w = m.group(0)
                    if lang == "en" and w == "I" and re.search(r'\bAct\s+I\b', s):
                        continue
                    filtered.append(w)
                if filtered:
                    print(f"{fpath}:{idx+1} {filtered}: {s}")
                    count += len(filtered)
    print(f"Total potential slips in {lang}: {count}\n")

def scan_th():
    print("=== SCANNING TH ===")
    count = 0
    for fpath in sorted(glob.glob("novels/locales/th/act_*/*.md")):
        lines = open(fpath, encoding="utf-8").readlines()
        for idx, line in enumerate(lines):
            s = line.strip()
            if is_dialogue_or_quote(s):
                continue
            narrative = re.sub(r'["“「].*?["”」]', '', s)
            matches = list(re.finditer(r'(ของฉัน|ฉัน)', narrative))
            if matches:
                filtered = [m.group(0) for m in matches]
                print(f"{fpath}:{idx+1} {filtered}: {s[:100]}")
                count += len(filtered)
    print(f"Total potential slips in th: {count}\n")

def scan_ja():
    print("=== SCANNING JA ===")
    count = 0
    for fpath in sorted(glob.glob("novels/locales/ja/act_*/*.md")):
        lines = open(fpath, encoding="utf-8").readlines()
        for idx, line in enumerate(lines):
            s = line.strip()
            if is_dialogue_or_quote(s):
                continue
            narrative = re.sub(r'["“「].*?["”」]', '', s)
            matches = list(re.finditer(r'(私の|私が|私を|私に|私|俺の|俺が|俺を|俺|僕の|僕が|僕を|僕)', narrative))
            if matches:
                filtered = [m.group(0) for m in matches]
                print(f"{fpath}:{idx+1} {filtered}: {s[:100]}")
                count += len(filtered)
    print(f"Total potential slips in ja: {count}\n")

def scan_ko():
    print("=== SCANNING KO ===")
    count = 0
    for fpath in sorted(glob.glob("novels/locales/ko/act_*/*.md")):
        lines = open(fpath, encoding="utf-8").readlines()
        for idx, line in enumerate(lines):
            s = line.strip()
            if is_dialogue_or_quote(s):
                continue
            narrative = re.sub(r'["“「].*?["”」]', '', s)
            matches = list(re.finditer(r'\b(나의|내가|나를|내|나|저의|제가|저를)\b', narrative))
            if matches:
                filtered = [m.group(0) for m in matches]
                print(f"{fpath}:{idx+1} {filtered}: {s[:100]}")
                count += len(filtered)
    print(f"Total potential slips in ko: {count}\n")

if __name__ == "__main__":
    scan_language("en", r"\b(I|my|me|mine|myself)\b")
    scan_language("es", r"\b(yo|mi|mis|conmigo)\b")
    scan_language("de", r"\b(ich|mich|mir|mein|meine|meinem|meinen|meiner|meines)\b")
    scan_language("ru", r"\b(я|меня|мне|мной|мною|мой|моя|моё|мое|мои|моего|моей|моих|моему|моим)\b")
    scan_language("zh", r"(?<![你他她])我(?!们)")
    scan_th()
    scan_ja()
    scan_ko()

