#!/usr/bin/env python3
"""
Mark AI revisions in VDP ver-2 files using <mark> tags.
Parses edited-note files, extracts corrections from ANY table format.

Token-efficient: shell script, not LLM processing.
"""

import re
import os
import sys

VER2_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
NOTES_DIR = os.path.join(VER2_DIR, 'edited-notes')
OUTPUT_DIR = os.path.join(VER2_DIR, 'marked')

# ---------------------------------------------------------------------------
# Myanmar translations for uncertainty notes
# Key: note_filename -> list of Myanmar notes
# ---------------------------------------------------------------------------
UNCERTAINTY_MM = {
    'vdp-001-005-note.md': [
        "**စာ ၂:** `နိဒါန်း အချိ` (မူရင်း OCR) → `နိဒါန်းအချီ` အဖြစ် ထားရှိသည် (နိဒါန်း၏ အမှန်တကယ်ခေါင်းစဉ်)။ အခြားစာမျက်နှာများရှိ ထပ်နေသော header များကဲ့သို့ မဖျက်ပါ။",
        "**စာ ၁:** `ဒုတိယကျမ်းခ` → `ဒုတိယကျမ်းစာ` (အတည်ပြုချက်အရ ပြင်ဆင်သည်)။",
        "**စာ ၃:** `မဟုတ်သေး,အခြေရ` – အနောက်တိုင်း ကော်မာ (`,`) ကို မြန်မာပုဒ်ဖြတ် (`၊`) အစား သုံးထားသည်။ မူရင်းစာအုပ်စာစီစာရိုက် ဖြစ်နိုင်သဖြင့် မပြင်ပါ။",
        "**စာ ၅:** `၉ ၅` နှင့် `သီတင်းကျွတ်လ` – ရက်စွဲဒေတာ (သီတင်းကျွတ်လ၊ ၉ ရက်၊ ၁၃၀၅ ခုနှစ် ဖြစ်နိုင်)။ metadata အဖြစ် ထိန်းသိမ်းထားသည်။",
        "**စာ ၅:** ဖိုင်အဆုံးရှိ `၁` – အောက်ခြေမှတ်စု အမှတ်အသား ဖြစ်နိုင်သည်။ မပြင်ပါ။",
        "**ပါဠိဂါထာ (စာ ၃):** `သဒ္ဒါ သုံဘုံ` – နိဿယအရ `သဒါသုဘံ` (sadāsubhaṃ)။ မူရင်းပါဠိကို မပြင်ပါ။",
    ],
    'vdp-006-010-note.md': [
        "**အရေးကြီးအမှား:** `လောဘပူစိတ်` → `လောဘမူစိတ်` (lobha-mūla)။ OCR က `မ` ကို `ပ` အဖြစ် မှားယွင်းဖတ်သည် — အဘိဓမ္မာ မသိလျှင် ရှာတွေ့ရန် အလွန်ခက်သော အမှားဖြစ်သည်။",
        "**နံပါတ် ၈ အမှား:** `ဂ` (ga) ကို `၈` (8) အစား စာမျက်နှာ ၁၀ တွင် \"စုစုပေါင်း ၈ စိတ်\" အကြောင်းအရာ၌ ၄ ကြိမ်တွေ့ရသည်။",
        "**ပါဠိအက္ခရာပျောက်:** `သမ္မာသမ္ဗုဒ္ဓမတုလံ` (Sammāsambuddhamatulaṃ) တွင် anusvara နှင့် stacked form ကြောင့် `လ` ပျောက်နေသည်။",
        "**မေးခွန်းများ (စာ ၈):** မေးခွန်းနံပါတ်များ OCR ကြောင့် ပျက်စီးနေသည်။ `က` မှ `တ` အထိ (၁၆ ခု) ပြန်လည်တည်ဆောက်ထားသည်။",
        "**စာ ၉-၁၀ ရှိ `စိတ်ပိုင်း`:** အဆင့်မြင့် အဘိဓမ္မာနည်းပညာပိုင်း ဖြစ်၍ ကျွမ်းကျင်သူမှ ပြန်လည်စစ်ဆေးရန် လိုအပ်သည်။",
    ],
    'vdp-086-090-note.md': [
        "**ပါဠိအရေးကြီး:** `အဂ္ဂမဂ္ဂဖဗုတေ` ကို နိဿယအရ (\"အရဟတ္တမဂ် အရဟတ္တဖိုလ်သည် ကြဉ်အပ်သော\") `အဂ္ဂမဂ္ဂဖလဌပိတေ` အဖြစ် ပြန်လည်တည်ဆောက်ထားသည်။ ယုံကြည်မှုမြင့်သော်လည်း [OCR_UNCERTAIN] ဟု မှတ်သားထားသည်။",
        "**\"စေတသိက် ကို ရေတွက်ခြင်း\" အပိုင်း:** အာရမ္မဏသင်္ဂဟမှ စေတသိက်သင်္ဂဟသို့ ကူးပြောင်းသည့်အပိုင်း။ စာ ၈၉-၉၀ အဆုံးရှိ ဆောင်ပုဒ်သည် စာမျက်နှာခွဲမှုကြောင့် ပြတ်တောက်နေရာ ပြန်ဆက်ထားသည်။",
        "**အထူး header:** စာ ၈၈ header `၁၉၆` (OCR အမှား: ၆→၉၊ အမှန်မှာ `၁၆၆`)။ မူရင်းစာအုပ်စာမျက်နှာ နံပါတ်များ (၁၆၃၊ ၁၆၇၊ ၁၆၉၊ ၁၇၁) ကို စာမျက်နှာအမှတ်အသားအဖြစ် ထားရှိသည်။",
        "**လင်္ကာပုံစံ:** လင်္ကာများ (ဧကန်လင်္ကာ၊ အနေကန်လင်္ကာ၊ ဆောင်ပုဒ်) ကို စုဏ္ဏိယနှင့် ခွဲခြားရန် blockquote ပုံစံဖြင့် ဖော်ပြထားသည်။",
        "**တိကျမှု ~၉၃%:** နေရာလွတ်နှင့် ပုဒ်ဖြတ်ပုဒ်ရပ်ဆိုင်ရာ အသေးစားအမှားအချို့ ကျန်ရှိနိုင်သည် (ပါဠိ + နိဿယ + မြန်မာရှင်းလင်းချက် + လင်္ကာ စသည့် ဘာသာစကားအလွှာများစွာ ရောနှောနေသောကြောင့်)။",
    ],
    'vdp-121-125-note.md': [
        "**Header `ပီထိပိုင်း` (စာ ၁၂၁-၁၂၂):** OCR က `ဝီထိ` ကို `ပီထိ` ဟု မှားယွင်းဖတ်သည်ဟု သံသယရှိသည်။ Header ဖြစ်၍ ဖျက်ထားပြီး အဓိကအကြောင်းအရာတွင် `ဝီထိ` မှန်ကန်စွာ သုံးထားသည်။",
        "**စာ ၁၂၄:** OCR က `၂၃၀` ဟုဖတ်သော်လည်း အစဉ်အရ (၂၃၂-၂၃၄-၂၃၆-၂၃၈-၂၄၀) `၂၃၈` ဖြစ်သင့်သည်။ Header ဖျက်ထားပြီးဖြစ်၍ အကြောင်းအရာကို မထိခိုက်ပါ။",
        "**စာ ၁၂၃ ဇယား:** `၆၆`, `ဘ`, `3`, `မ`, `000` တို့သည် မဂ္ဂဝီထိဖော်ပြချက် ဇယား၏ အကြွင်းအကျန်များဖြစ်သည်။ ရိုးရှင်းသော စာသားဖော်ပြချက်ဖြင့် အစားထိုးထားသည်။",
        "**ပြန့်ကျဲနေသော နံပါတ်များ:** `၉`, `8`, `၁`, `၁၅` တို့သည် ပါဠိတွင် ပြန့်ကျဲနေသည် — OCR စာလုံးမှားယွင်းဖတ်သည့် ပုံစံ။",
        "**စာ ၁၂၅:** `ပုဂ္ဂလဘေဒ` အပိုင်းအသစ် စတင်သည်၊ အဖွင့်အပိုင်းသာရှိသေးသည်။",
    ],
    'vdp-131-135-note.md': [
        "**စာ ၁၃၂:** မူရင်းတွင် ဇယားပါရှိသည် — ဖတ်ရှုရလွယ်ကူစေရန် Markdown ဇယားအဖြစ် ပြောင်းထားသည်။ `55-10cc`, `0` တို့ကို ဖယ်ရှားထားသည်။",
        "**စာ ၁၃၃:** မေးခွန်းများကို က မှ လ အထိ (၂၈ ခု) နံပါတ်တပ်ထားသည်။ OCR မှားယွင်းဖတ်ထားသော စာလုံးအချို့ကို ပြင်ဆင်ထားသည် (C→ဂ၊ ၈→ဂ စသည်)။",
        "**စာ ၁၃၄:** ပထမ ပါဠိစာပိုဒ် ဆိုးရွားစွာ ပျက်စီးနေသည် — နိဿယအကြောင်းအရာကို အခြေခံ၍ ပြန်လည်တည်ဆောက်ထားသည်။ ဖြစ်နိုင်ပါက မူရင်းနှင့် ပြန်လည်စစ်ဆေးရန် လိုအပ်သည်။",
        "**စာ ၁၃၅:** `ဘီလာ` → မပြင်ပါ (`ဖီလာ` ဖြစ်နိုင်သော်လည်း မသေချာ — ပါဠိ `တိရော` ၏ ရှင်းလင်းချက်ဖြစ်သည်)။",
        "**စုစုပေါင်း ပြင်ဆင်ထားသော အမှား: ~၅၅ ခု** (header/artifact အပါအဝင်)",
    ],
}

# ---------------------------------------------------------------------------
# Flexible parser — handles all note file formats
# ---------------------------------------------------------------------------

def parse_corrections(note_path):
    """
    Extract all corrections from a note file regardless of table format.
    Returns list of corrected_text strings.
    """
    corrections = []
    with open(note_path, 'r', encoding='utf-8') as f:
        text = f.read()

    lines = text.split('\n')
    in_table = False
    sua_col = -1
    loi_col = -1
    header_seen = False

    for line in lines:
        stripped = line.strip()

        if ('lỗi' in stripped.lower() or 'bảng lỗi' in stripped.lower()) \
           and (stripped.startswith('##') or stripped.startswith('###')):
            in_table = True
            sua_col = -1
            loi_col = -1
            header_seen = False
            continue

        if stripped.startswith('##') and 'lưu ý' in stripped.lower():
            in_table = False
            continue
        if stripped.startswith('##') and 'header' in stripped.lower():
            in_table = False
            continue
        if stripped.startswith('###') and ('lưu ý' in stripped.lower() or 'header' in stripped.lower()):
            in_table = False
            continue

        if not in_table:
            continue

        if '|---' in line:
            continue

        if line.startswith('|') and ('Lỗi' in line or 'Sửa' in line or 'Sau' in line):
            cells = [c.strip() for c in line.split('|')]
            for i, cell in enumerate(cells):
                if cell in ('Sửa', 'Đã sửa', 'Sau', 'Sửa thành'):
                    sua_col = i
                if cell in ('Lỗi', 'Lỗi gốc', 'Trước'):
                    loi_col = i
            header_seen = True
            continue

        if line.startswith('|') and sua_col >= 0 and header_seen:
            cells = [c.strip() for c in line.split('|')]
            if sua_col < len(cells):
                sua_text = cells[sua_col]

                if sua_text.upper() in ('XÓA', 'XOA', 'XOÁ', 'DELETE', '—', '-', ''):
                    continue
                if 'XÓA' in sua_text.upper() and '`' not in sua_text:
                    continue

                bt_matches = re.findall(r'`([^`]+)`', sua_text)
                for match in bt_matches:
                    match = match.strip()
                    if len(match) >= 2 and not re.match(r'^[0-9\s\.\-\'"]+$', match):
                        corrections.append(match)

    # Also parse list-format corrections (e.g., `error` → `corrected`)
    for match in re.finditer(r'`([^`]+)`\s*[→]\s*`([^`]+)`', text):
        corrected = match.group(2).strip()
        if corrected and len(corrected) >= 2 \
           and not re.match(r'^[0-9\s\.\-\'"]+$', corrected) \
           and corrected.upper() not in ('XÓA', 'XOA', 'XOÁ'):
            corrections.append(corrected)

    return corrections


def get_uncertainty_notes_mm(note_filename):
    """Return Myanmar-translated uncertainty notes for a given note file."""
    return UNCERTAINTY_MM.get(note_filename, [])


# ---------------------------------------------------------------------------
# Marking engine
# ---------------------------------------------------------------------------

def apply_mark_tags(content, corrections):
    """Wrap each corrected segment in <mark> tags, longest-first, no nesting."""
    unique = list(set(corrections))
    unique.sort(key=len, reverse=True)

    pmap = {}
    for i, corr in enumerate(unique):
        tok = f'\x00MK{i}\x01'
        pmap[tok] = corr
        limit = 3 if len(corr) <= 2 else -1

        parts = []
        last = 0
        parts_count = 0
        for m in re.finditer(re.escape(corr), content):
            if limit > 0 and parts_count >= limit:
                break
            before = content[max(0, m.start()-10):m.start()]
            after = content[m.end():m.end()+10]
            if '\x00' in before or '\x00' in after:
                continue
            parts.append(content[last:m.start()])
            parts.append(tok)
            last = m.end()
            parts_count += 1
        parts.append(content[last:])
        content = ''.join(parts)

    for tok, corr in pmap.items():
        content = content.replace(tok, f'<mark>{corr}</mark>')

    return content


# ---------------------------------------------------------------------------
# Main
# ---------------------------------------------------------------------------

def process_batch(page_files=None):
    os.makedirs(OUTPUT_DIR, exist_ok=True)

    if page_files is None:
        page_files = sorted([
            f for f in os.listdir(VER2_DIR)
            if f.startswith('vdp-') and f.endswith('.md') and '-note' not in f
        ])

    stats = []
    for page_file in page_files:
        base = page_file.replace('.md', '')
        note_file = f'{base}-note.md'
        note_path = os.path.join(NOTES_DIR, note_file)

        if not os.path.exists(note_path):
            page_path = os.path.join(VER2_DIR, page_file)
            with open(page_path, 'r', encoding='utf-8') as f:
                content = f.read()
            output_path = os.path.join(OUTPUT_DIR, page_file)
            with open(output_path, 'w', encoding='utf-8') as f:
                f.write(content)
            print(f"  ⚠ {page_file}: no note → copy as-is")
            continue

        corrections = parse_corrections(note_path)
        uncertainties = get_uncertainty_notes_mm(note_file)

        page_path = os.path.join(VER2_DIR, page_file)
        with open(page_path, 'r', encoding='utf-8') as f:
            content = f.read()

        if corrections:
            content = apply_mark_tags(content, corrections)

        if uncertainties:
            content += '\n\n---\n\n## ❓ မေးခွန်းထုတ်စရာများနှင့် အထူးမှတ်ချက်များ\n\n'
            for i, note in enumerate(uncertainties, 1):
                content += f'{i}. {note}\n'

        output_path = os.path.join(OUTPUT_DIR, page_file)
        with open(output_path, 'w', encoding='utf-8') as f:
            f.write(content)

        marked_count = content.count('<mark>')
        stats.append({
            'file': page_file,
            'corrections': len(corrections),
            'uncertainties': len(uncertainties),
            'marks': marked_count,
        })
        flag = '✓' if corrections else '→'
        uflag = f' +{len(uncertainties)}⚠' if uncertainties else ''
        print(f"  {flag} {page_file}: {len(corrections)} edits → ~{marked_count} marks{uflag}")

    total_c = sum(s['corrections'] for s in stats)
    total_u = sum(s['uncertainties'] for s in stats)
    total_m = sum(s['marks'] for s in stats)
    print(f"\n{'='*50}")
    print(f"Total: {len(stats)} files, {total_c} corrections, {total_u} uncertainties, ~{total_m} marks")
    print(f"Output: {OUTPUT_DIR}/")
    return stats


if __name__ == '__main__':
    import argparse
    ap = argparse.ArgumentParser()
    ap.add_argument('--files', nargs='*')
    ap.add_argument('--limit', type=int, default=0)
    args = ap.parse_args()

    files = args.files if args.files else None
    if files is None and args.limit > 0:
        all_files = sorted([
            f for f in os.listdir(VER2_DIR)
            if f.startswith('vdp-') and f.endswith('.md') and '-note' not in f
        ])
        files = all_files[:args.limit]

    process_batch(files)
