"""Read one electrical one-line diagram image into JSON. python usage.py drawing.png [--device cuda|cpu] Apache-2.0, Copyright 2026 Decosa. """ import argparse import json import re import torch from PIL import Image from transformers import AutoModelForImageTextToText, AutoProcessor REPO = "decosaai/decosa-oneline-reader-paddleocr-vl" PROMPT = "One-line Diagram Recognition:" MAX_PIXELS = 1003520 # the processor's cap def fit(im: Image.Image) -> Image.Image: """Scale to just under the processor's pixel cap; small or poor scans are enlarged (as in training).""" w, h = im.size s = (MAX_PIXELS * 0.98 / (w * h)) ** 0.5 if s < 1 or w < 1100: im = im.resize((max(28, int(w * s)), max(28, int(h * s))), Image.BICUBIC) return im def main() -> None: ap = argparse.ArgumentParser() ap.add_argument("image") ap.add_argument("--model", default=REPO) ap.add_argument("--device", default="cuda" if torch.cuda.is_available() else "cpu") a = ap.parse_args() proc = AutoProcessor.from_pretrained(a.model) dtype = torch.bfloat16 if a.device == "cuda" else torch.float32 model = AutoModelForImageTextToText.from_pretrained(a.model, dtype=dtype).to(a.device).eval() im = fit(Image.open(a.image).convert("L")).convert("RGB") msgs = [{"role": "user", "content": [{"type": "image", "image": im}, {"type": "text", "text": PROMPT}]}] x = proc.apply_chat_template(msgs, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt").to(a.device) with torch.inference_mode(): out = model.generate(**x, max_new_tokens=900, do_sample=False, use_cache=True) text = proc.decode(out[0][x["input_ids"].shape[-1]:], skip_special_tokens=True) text = re.sub(r"<\|[a-z_]+\|>|||", "", text).strip() try: print(json.dumps(json.loads(text), indent=1)) except json.JSONDecodeError: print(text) if __name__ == "__main__": main()