from pathlib import Path
from hashlib import sha256
from zipfile import ZipFile
from shutil import copy2
import json, re
from docx import Document
from pypdf import PdfReader
from PIL import Image

root=Path(r'C:\Users\Asus\Desktop\TourAppProject\travelin_app\docs\thesis_appendices')
state=json.loads((root/'work/profile_dashboard_batch.json').read_text(encoding='utf-8'))
word=root/'TourApp_Appendix_A_B.docx';pdf=root/'TourApp_Appendix_A_B_Review.pdf'
doc=Document(word)
assert len(doc.inline_shapes)==16
assert sum('[รอภาพหน้าจอจริง:' in p.text for p in doc.paragraphs)==33
headings=[p.text for p in doc.paragraphs if p.style.name=='Heading 2']
assert len(headings)==54
caps=[p.text for p in doc.paragraphs if re.match(r'ภาพประกอบที่ [กข]-\d+ ',p.text)]
assert [re.match(r'ภาพประกอบที่ (ข-\d+) ',p).group(1) for p in caps if ' ข-' in p]==[f'ข-{i}' for i in range(1,12)]
with ZipFile(word) as z:
    assert z.testzip() is None
    hashes={sha256(z.read(n)).hexdigest() for n in z.namelist() if n.startswith('word/media/')}
    for output in state['outputs']:
        assert sha256(Path(output).read_bytes()).hexdigest() in hashes
    for source in state['sources']:
        assert sha256(Path(source).read_bytes()).hexdigest() not in hashes
    assert len([n for n in z.namelist() if n.startswith('word/media/')])==15
assert len(PdfReader(pdf).pages)==34
pending=(root/'appendix_pending.md').read_text(encoding='utf-8')
assert len([l for l in pending.splitlines() if re.match(r'\| [กข]\.\d',l)])==33
state.update({'pdf_pages':34,'topics':54,'unique_media':15,
    'sanitized_media_embedded':True,'raw_originals_not_embedded':True,
    'zip_crc':'PASS','visual_review':'15 changed pages inspected; 19 pixel-identical to previously reviewed pages',
    'image_pages':{'tourist_profile':14,'agency_dashboard':27},
    'word_sha256':sha256(word.read_bytes()).hexdigest(),'pdf_sha256':sha256(pdf.read_bytes()).hexdigest(),
    'source_code_changed':False,'database_operations':0,'ui_business_operations':0})
cp=Path(state['checkpoint'])
for name in (word.name,pdf.name,'appendix_progress.md','appendix_pending.md','appendix_image_index.md','appendix_capture_guide.md'):
    copy2(root/name,cp/name)
(cp/'verification.json').write_text(json.dumps(state,ensure_ascii=False,indent=2),encoding='utf-8')
print(json.dumps({k:state[k] for k in ['checkpoint','embedded_images','unique_media','pending_topics','topics','pdf_pages','zip_crc','sanitized_media_embedded','raw_originals_not_embedded','word_sha256','pdf_sha256']},ensure_ascii=False))
