"""Fix only cross-reference styling in the saved appendix; never rebuild it."""
from datetime import datetime
from hashlib import sha256
from pathlib import Path
from shutil import copy2
from zipfile import ZipFile
import json
import re

from docx import Document

ROOT = Path(r"C:\Users\Asus\Desktop\TourAppProject\travelin_app\docs\thesis_appendices")
WORD = ROOT / "TourApp_Appendix_A_B.docx"
stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
checkpoint = ROOT / "checkpoints" / f"verified_reference_fix_{stamp}"
checkpoint.mkdir(parents=True, exist_ok=False)
for name in (WORD.name, "TourApp_Appendix_A_B_Review.pdf", "appendix_progress.md",
             "appendix_pending.md", "appendix_image_index.md"):
    source = ROOT / name
    if source.exists():
        copy2(source, checkpoint / ("before_" + name))

doc = Document(WORD)
changed = 0
for paragraph in doc.paragraphs:
    if paragraph.style.name != "Caption":
        continue
    text = paragraph.text
    if text.startswith("ภาพประกอบที่ ข-1 แสดงหน้าแรก") and "ใช้ภาพร่วม" in text:
        # Retain the existing italic/font/paragraph formatting of its runs.
        paragraph.runs[0].text = "ใช้" + paragraph.runs[0].text
        paragraph.style = doc.styles["Normal"]
        changed += 1
    elif text.startswith("ใช้ภาพประกอบที่ "):
        paragraph.style = doc.styles["Normal"]
        changed += 1

staging = ROOT / "work" / "appendix_reference_style_fix.docx"
doc.save(staging)
check = Document(staging)
ids = [re.match(r"ภาพประกอบที่ ([กข]-\d+) ", p.text).group(1)
       for p in check.paragraphs
       if p.style.name == "Caption" and re.match(r"ภาพประกอบที่ ([กข]-\d+) ", p.text)]
assert len(ids) == 14 and len(set(ids)) == 14, "Caption count/uniqueness changed"
assert len(check.inline_shapes) == 14, "Image count changed"
assert sum("[รอภาพหน้าจอจริง:" in p.text for p in check.paragraphs) == 35, "Pending count changed"
with ZipFile(staging) as package:
    assert package.testzip() is None, "DOCX ZIP CRC failed"
staging.replace(WORD)
copy2(WORD, checkpoint / WORD.name)
result = {"checkpoint": str(checkpoint), "reference_styles_fixed": changed,
          "actual_caption_count": len(ids), "caption_ids_unique": True,
          "embedded_image_count": 14, "pending_topics": 35,
          "word_sha256": sha256(WORD.read_bytes()).hexdigest()}
(checkpoint / "verification.json").write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding="utf-8")
print(json.dumps(result, ensure_ascii=False))
