"""Create owned teaching fixtures, parse PDF and run pinned public encoders on CPU."""
import argparse
from datetime import datetime, timezone
import hashlib
import json
from pathlib import Path
import platform
import time

ROOT = Path(__file__).resolve().parent


def save(name, value):
    (ROOT / name).write_text(json.dumps(value, ensure_ascii=False, indent=2) + '\n', encoding='utf8')


def fixtures():
    from reportlab.pdfgen.canvas import Canvas
    from PIL import Image, ImageDraw, ImageFont
    import pdfplumber
    from pypdf import PdfReader
    import pypdfium2
    manual = ROOT / 'synthetic-manual.pdf'
    canvas = Canvas(str(manual), pagesize=(595, 842), invariant=1)
    rows = [[['Model', 'Supply', 'Tank'], ['AX-220', '220 V', '20 L'], ['AX-110', '110 V', '10 L']],
            [['Model', 'Use', 'Restriction'], ['AX-220', 'Indoor', 'Not outdoor'], ['AX-110', 'Indoor', 'Not wet areas']]]
    for page, table in enumerate(rows, 1):
        canvas.setTitle('Synthetic B2B manual: table extraction fixture')
        canvas.setFont('Helvetica-Bold', 19)
        canvas.drawString(48, 784, 'Synthetic B2B Manual')
        canvas.setFont('Helvetica', 11)
        canvas.drawString(48, 758, 'Teaching fixture only. No real customer product. CC0-1.0.')
        for x in [48, 200, 350, 548]: canvas.line(x, 610, x, 730)
        for y in [610, 650, 690, 730]: canvas.line(48,y,548,y)
        # Deliberately valid column-major content stream: visual rows are unchanged.
        for col in [2, 0, 1]:
            for row in range(3):
                canvas.setFont('Helvetica-Bold' if row==0 else 'Helvetica',11)
                canvas.drawString([58,210,360][col],704-row*40,table[row][col])
        canvas.drawString(48, 576, 'Values apply only to the model and conditions in the same row.')
        canvas.drawString(48, 558, 'Do not infer other voltages, certifications or warranty coverage.')
        canvas.drawString(48, 48, f'Fixture version 1 | Page {page} of 2')
        canvas.showPage()
    canvas.save()
    reader = PdfReader(manual)
    texts = [p.extract_text() for p in reader.pages]
    with pdfplumber.open(manual) as pdf:
        tables = [p.extract_table() for p in pdf.pages]
        words = [p.extract_words() for p in pdf.pages]
    rendered = pypdfium2.PdfDocument(manual)
    for i in range(len(rendered)):
        rendered[i].render(scale=1.5).to_pil().save(ROOT / f'manual-page-{i+1}.png')
    save('document-results.json', {'kind':'owned-synthetic-fixture','pages':len(reader.pages),
         'source_sha256':hashlib.sha256(manual.read_bytes()).hexdigest(),
         'pypdf':{'plain_text':texts}, 'pdfplumber':{'tables':tables,'word_coordinates':words},
         'expected_tables':rows, 'table_exact_match':[a==b for a,b in zip(tables,rows)],
         'interpretation':'Plain text loses explicit cell associations; table extraction is compared to the authored table, not to an LLM.'})
    image_rows = [('red-circle-220','red','circle','AX-220'),('red-circle-110','red','circle','AX-110'),
                  ('blue-square-220','blue','square','BX-220'),('green-triangle-110','green','triangle','CX-110')]
    font=ImageFont.load_default(size=24)
    for key,color,shape,model in image_rows:
        image=Image.new('RGB',(384,384),'white')
        draw=ImageDraw.Draw(image)
        if shape=='circle': draw.ellipse((72,60,312,300),fill=color)
        if shape=='square': draw.rectangle((72,60,312,300),fill=color)
        if shape=='triangle': draw.polygon([(192,60),(312,300),(72,300)],fill=color)
        draw.text((20,336),f'{model} | teaching fixture',font=font,fill='black')
        image.save(ROOT / f'{key}.png')
    facts=[
        ('D1','AX-220 uses 220 V; tank 20 L; indoor sealed hard floors only; not outdoors.','AX-220使用220 V供电，水箱20 L；仅限室内密封硬地面，不可用于户外。'),
        ('D2','AX-110 uses 110 V; tank 10 L; indoor only; not wet areas.','AX-110使用110 V供电，水箱10 L；仅限室内，不适用于潮湿区域。'),
        ('D3','BX-220 red cable is 5 m; connector type K; not compatible with type J.','BX-220红色电缆长5 m，K型接头，与J型不兼容。'),
        ('D4','BX-110 blue cable is 3 m; connector type J.','BX-110蓝色电缆长3 m，J型接头。'),
        ('D5','CZ-10 replacement pad is for sealed stone; do not use on untreated wood.','CZ-10替换垫适用于密封石材，不可用于未经处理的木地板。'),
        ('D6','CZ-11 replacement pad is for untreated wood; dry use only.','CZ-11替换垫适用于未经处理的木地板，仅可干用。'),
        ('D7','DW-1 carton contains 12 pieces; MOQ is 4 cartons.','DW-1每箱12件，最小订购量为4箱。'),
        ('D8','DW-2 carton contains 24 pieces; MOQ is 2 cartons.','DW-2每箱24件，最小订购量为2箱。'),
        ('D9','EV-1 weighs 8 kg net and 10 kg packed; freight calculations use packed weight.','EV-1净重8 kg，包装重10 kg，运输计算使用包装重。'),
        ('D10','EV-2 weighs 10 kg net and 13 kg packed.','EV-2净重10 kg，包装重13 kg。'),
        ('D11','FX-1 supply is 24 V DC; laboratory sample, not a certification statement.','FX-1供电24 V直流，属于实验室样品，不构成认证声明。'),
        ('D12','FX-2 supply is 220 V AC; certification status is not supplied.','FX-2供电220 V交流，未提供认证状态。')]
    queries=[
        ('Q1','Which voltage does AX-220 use?','AX-220使用什么电压？',['D1']),
        ('Q2','Can AX-110 be used in wet areas?','AX-110能在潮湿区域使用吗？',['D2']),
        ('Q3','Which cable has a K connector and is five metres long?','哪种电缆为K型接头、长五米？',['D3']),
        ('Q4','Which cable uses a J connector?','哪种电缆使用J型接头？',['D4']),
        ('Q5','Which pad must not be used on untreated wood?','哪种垫不能用于未经处理的木地板？',['D5']),
        ('Q6','Which pad is dry-use only?','哪种垫只能干用？',['D6']),
        ('Q7','How many pieces are in a DW-1 carton?','DW-1每箱多少件？',['D7']),
        ('Q8','What is the minimum order for DW-2?','DW-2最小订购量是多少？',['D8']),
        ('Q9','Which EV-1 weight should be used for freight?','EV-1运输应使用哪个重量？',['D9']),
        ('Q10','What is the packed weight of EV-2?','EV-2的包装重是多少？',['D10']),
        ('Q11','What is the warranty period for FX-1?','FX-1质保期多长？',[]),
        ('Q12','Is FX-2 CE certified?','FX-2是否获得CE认证？',[])]
    save('benchmark.json',{'version':'2.0.0','license':'CC0-1.0','kind':'synthetic-teaching',
        'labels':'author-compiled; no independent human adjudication',
        'documents':[{'id':i,'en':en,'zh':zh} for i,en,zh in facts],
        'queries':[{'id':i,'en':en,'zh':zh,'relevant':rel} for i,en,zh,rel in queries],
        'images':[{'id':i,'color':c,'shape':s,'model':m,'file':i+'.png'} for i,c,s,m in image_rows]})


def models():
    import numpy as np
    import torch
    from PIL import Image
    from huggingface_hub import model_info
    from sentence_transformers import SentenceTransformer
    from transformers import CLIPModel, CLIPProcessor
    data=json.loads((ROOT/'benchmark.json').read_text())
    definitions={'multilingual':'sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2',
                 'image_text':'openai/clip-vit-base-patch32'}
    lock_path=ROOT/'model-revisions.json'
    if lock_path.exists(): lock=json.loads(lock_path.read_text())
    else:
        lock={k:{'id':name,'revision':model_info(name).sha} for k,name in definitions.items()}
        save('model-revisions.json',lock)
    start=time.monotonic()
    embedding=SentenceTransformer(lock['multilingual']['id'], revision=lock['multilingual']['revision'], device='cpu')
    cross=[]
    for corpus_lang in ['zh','en']:
        doc_vectors=embedding.encode([d[corpus_lang] for d in data['documents']], normalize_embeddings=True)
        for query_lang in ['zh','en']:
            query_vectors=embedding.encode([q[query_lang] for q in data['queries']], normalize_embeddings=True)
            matrix=query_vectors@doc_vectors.T
            for q,scores in zip(data['queries'],matrix):
                ranking=sorted(range(len(scores)),key=lambda i:(-float(scores[i]),data['documents'][i]['id']))
                top=[{'id':data['documents'][i]['id'],'score':float(scores[i])} for i in ranking[:3]]
                relevant=set(q['relevant'])
                cross.append({'query_id':q['id'],'query_language':query_lang,'corpus_language':corpus_lang,
                    'top3':top,'top1_correct':top[0]['id'] in relevant if relevant else None,
                    'recall_at_3':len({d['id'] for d in top}&relevant)/len(relevant) if relevant else None,
                    'answerability':'has-source' if relevant else 'no-answer-key'})
    del embedding
    clip=CLIPModel.from_pretrained(lock['image_text']['id'],revision=lock['image_text']['revision']).eval()
    processor=CLIPProcessor.from_pretrained(lock['image_text']['id'],revision=lock['image_text']['revision'])
    prompts=['a red circle','a blue square','a green triangle','AX-220 red circle','AX-110 red circle']
    inputs=processor(text=prompts,images=[Image.open(ROOT/i['file']) for i in data['images']],return_tensors='pt',padding=True)
    with torch.no_grad(): scores=clip(**inputs).logits_per_text.numpy()
    image_text=[]
    for prompt,row in zip(prompts,scores):
        ranking=sorted(range(len(row)),key=lambda i:-float(row[i]))
        image_text.append({'query':prompt,'ranking':[{'id':data['images'][i]['id'],'logit':float(row[i])} for i in ranking]})
    save('model-results.json',{'data_kind':'synthetic-teaching','external_ai_tested':False,
         'independent_human_review':False,'models':lock,'device':'cpu',
         'generated_at':datetime.now(timezone.utc).isoformat(),'elapsed_seconds':time.monotonic()-start,
         'python':platform.python_version(),'cross_language':cross,'image_text':image_text,
         'limits':'Tiny author-labeled corpus, no tuning/held-out claim. CLIP similarity is not product compatibility or confidence.'})
    print(json.dumps({'cross_language_rows':len(cross),'image_text_rows':len(image_text),'elapsed_seconds':time.monotonic()-start}))


if __name__=='__main__':
    parser=argparse.ArgumentParser()
    parser.add_argument('--models',action='store_true')
    args=parser.parse_args()
    if args.models: models()
    else: fixtures()
