Image-Text-to-Text
MLX
Safetensors
unlimited-ocr
ax-engine
mlx-vlm
ocr
mxfp8
int8
apple-silicon
automatosx
conversational
8-bit precision
Instructions to use AutomatosX/AX-Unlimited-OCR-3B-MoE-MLX-MXFP8 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use AutomatosX/AX-Unlimited-OCR-3B-MoE-MLX-MXFP8 with MLX:
# Make sure mlx-vlm is installed # pip install --upgrade mlx-vlm from mlx_vlm import load, generate from mlx_vlm.prompt_utils import apply_chat_template from mlx_vlm.utils import load_config # Load the model model, processor = load("AutomatosX/AX-Unlimited-OCR-3B-MoE-MLX-MXFP8") config = load_config("AutomatosX/AX-Unlimited-OCR-3B-MoE-MLX-MXFP8") # Prepare input image = ["http://images.cocodataset.org/val2017/000000039769.jpg"] prompt = "Describe this image." # Apply chat template formatted_prompt = apply_chat_template( processor, config, prompt, num_images=1 ) # Generate output output = generate(model, processor, formatted_prompt, image) print(output) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Atomic Chat
| """Normalize OCR output for fair comparison. | |
| Strips model-specific artifacts, normalizes whitespace, and removes | |
| special tokens before evaluation. | |
| Usage: | |
| python benchmarks/normalize_output.py input.txt output.txt | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import re | |
| import unicodedata | |
| from pathlib import Path | |
| def normalize_ocr_output(text: str) -> str: | |
| """Normalize OCR output for evaluation. | |
| Steps: | |
| 1. Remove special tokens (<|det|>, <|/det|>, <|grounding|>, etc.) | |
| 2. Unicode NFC normalization | |
| 3. Normalize whitespace (collapse multiple spaces/newlines) | |
| 4. Strip leading/trailing whitespace | |
| 5. Normalize common OCR artifacts | |
| """ | |
| if not isinstance(text, str): | |
| raise TypeError("text must be a string") | |
| # Tokenizer-visible whitespace markers occasionally survive decoding. | |
| text = text.replace("Ġ", " ").replace("Ċ", "\n") | |
| # Remove grounding coordinates only when they are part of a detection | |
| # span. A blanket ``[digits]`` removal corrupts legitimate text such as | |
| # citations and list markers. | |
| number = r"-?\d+(?:\.\d+)?" | |
| coordinate_pattern = re.compile( | |
| rf"\[\s*{number}\s*,\s*{number}\s*,\s*{number}\s*,\s*{number}\s*\]" | |
| ) | |
| # In the older ref/det form, the ref span is recognized content and should | |
| # be retained while the following coordinate-only det span is removed. | |
| text = re.sub( | |
| r"<\|ref\|>(.*?)<\|/ref\|>\s*<\|det\|>.*?<\|/det\|>", | |
| lambda match: match.group(1), | |
| text, | |
| flags=re.DOTALL, | |
| ) | |
| def clean_detection(match: re.Match) -> str: | |
| inner = match.group(1) | |
| # Current output uses ``<|det|>layout-label [box]<|/det|>content``. | |
| # Both the layout label and coordinates are metadata, not OCR text. | |
| return "" if coordinate_pattern.search(inner) else inner | |
| text = re.sub( | |
| r"<\|det\|>(.*?)<\|/det\|>", | |
| clean_detection, | |
| text, | |
| flags=re.DOTALL, | |
| ) | |
| # Some decoded outputs omit the first opening <|det|> token. | |
| text = re.sub( | |
| rf"(?m)(^|\n)[^\n<]*?{coordinate_pattern.pattern}" | |
| rf"(?=\s*<\|/det\|>)", | |
| lambda match: match.group(1), | |
| text, | |
| ) | |
| # Remove special tokens | |
| text = re.sub(r"<\|[^|]*\|>", "", text) | |
| text = re.sub(r"<|[^|]*|>", "", text) | |
| text = re.sub(r"<PAGE>", "\n", text, flags=re.IGNORECASE) | |
| text = re.sub(r"<image>", "", text) | |
| # Unicode normalization | |
| text = unicodedata.normalize("NFC", text) | |
| # Normalize different types of whitespace | |
| text = text.replace("\r\n", "\n").replace("\r", "\n") | |
| # Collapse multiple blank lines to max 2 | |
| text = re.sub(r"\n{3,}", "\n\n", text) | |
| # Collapse multiple spaces (but preserve single newlines) | |
| text = re.sub(r"[^\S\n]+", " ", text) | |
| # Strip each line | |
| lines = [line.strip() for line in text.split("\n")] | |
| text = "\n".join(lines) | |
| # Whitespace-only lines become empty only after the per-line strip above, | |
| # so enforce the blank-line limit again afterwards. | |
| text = re.sub(r"\n{3,}", "\n\n", text) | |
| # Strip overall | |
| text = text.strip() | |
| return text | |
| def normalize_for_digit_comparison(text: str) -> str: | |
| """Extract only digits and decimal points for numeric comparison.""" | |
| return re.sub(r"[^0-9.]", "", text) | |
| def main(): | |
| parser = argparse.ArgumentParser(description="Normalize OCR output for evaluation") | |
| parser.add_argument("input", type=Path, help="Raw OCR text file") | |
| parser.add_argument("output", type=Path, help="Normalized output file") | |
| args = parser.parse_args() | |
| input_path = args.input | |
| output_path = args.output | |
| text = input_path.read_text(encoding="utf-8") | |
| normalized = normalize_ocr_output(text) | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| output_path.write_text(normalized, encoding="utf-8") | |
| print(f"Normalized: {input_path} -> {output_path}") | |
| print(f" Input: {len(text)} chars") | |
| print(f" Output: {len(normalized)} chars") | |
| if __name__ == "__main__": | |
| main() | |