"""Example: OCR a multi-page PDF document.""" from unlimited_ocr import OCRPipeline # Initialize the pipeline pipeline = OCRPipeline( model_path="AutomatosX/AX-Unlimited-OCR-3B-MoE-MLX-MXFP8", verbose=True, ) # --- Basic PDF OCR (all pages) --- result = pipeline.run("your_document.pdf", format="text", dpi=300) print(result) # --- Save as Markdown with page headings --- result = pipeline.run( "your_document.pdf", format="markdown", dpi=300, output_path="output.md", ) print("Saved to output.md") # --- JSON output with per-page structure --- result = pipeline.run( "your_document.pdf", format="json", dpi=300, output_path="output.json", ) print("Saved to output.json") # --- With preprocessing for scanned PDFs --- result = pipeline.run( "scanned_document.pdf", format="text", preprocess=True, # deskew + contrast enhancement per page dpi=300, ) print(result) # --- With progress tracking --- def on_progress(current, total): print(f" Processing page {current}/{total}...") result = pipeline.run( "long_document.pdf", format="text", progress_callback=on_progress, ) # Clean up pipeline.cleanup()