"""PDF/image loading and OCR. Native (embedded) text is pulled straight out of the PDF when present; pages without a usable text layer (scanned documents) are rendered to an image and run through Tesseract instead. """ import os import fitz # PyMuPDF import pytesseract from PIL import Image MIN_EMBEDDED_TEXT_LEN = 20 RENDER_DPI = 300 IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".tif", ".tiff", ".bmp"} def _ocr_image(img): data = pytesseract.image_to_data(img, output_type=pytesseract.Output.DICT) confidences = [int(c) for c in data.get("conf", []) if str(c).isdigit() and int(c) >= 0] mean_conf = round(sum(confidences) / len(confidences), 1) if confidences else None text = pytesseract.image_to_string(img) return text, mean_conf def _process_pdf(file_path): pages = [] doc = fitz.open(file_path) try: for page_num, page in enumerate(doc, start=1): embedded_text = page.get_text().strip() if len(embedded_text) >= MIN_EMBEDDED_TEXT_LEN: pages.append({ "page": page_num, "text": embedded_text, "source": "embedded", "ocr_confidence": None, }) continue pix = page.get_pixmap(dpi=RENDER_DPI) img = Image.frombytes("RGB", [pix.width, pix.height], pix.samples) ocr_text, mean_conf = _ocr_image(img) pages.append({ "page": page_num, "text": ocr_text.strip(), "source": "ocr", "ocr_confidence": mean_conf, }) finally: doc.close() return pages def _process_image(file_path): img = Image.open(file_path).convert("RGB") text, mean_conf = _ocr_image(img) return [{ "page": 1, "text": text.strip(), "source": "ocr", "ocr_confidence": mean_conf, }] def extract_pages(file_path): """Return a list of {page, text, source, ocr_confidence} dicts for a file.""" ext = os.path.splitext(file_path)[1].lower() if ext == ".pdf": return _process_pdf(file_path) if ext in IMAGE_EXTENSIONS: return _process_image(file_path) raise ValueError(f"Unsupported file type: {ext}")