Record public Phoenix release status
Browse files- .gitattributes +0 -76
- PUBLISH_CHECKLIST.md +4 -4
- README.md +233 -66
- SHA256SUMS +1 -1
- metadata.json +1 -1
.gitattributes
CHANGED
|
@@ -1,78 +1,2 @@
|
|
| 1 |
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 2 |
|
| 3 |
-
presentation_dashboard/assets/before_after/manuscript.png filter=lfs diff=lfs merge=lfs -text
|
| 4 |
-
presentation_dashboard/assets/samples/clear/manuscript.jpg filter=lfs diff=lfs merge=lfs -text
|
| 5 |
-
presentation_dashboard/assets/samples/logic/manuscript.png filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
presentation_dashboard/assets/samples/medium/manuscript.jpg filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
reports/تسليم_صباحي/لقطات/01_تسجيل_الدخول.png filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
reports/تسليم_صباحي/لقطات/02_التفريغ.png filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
reports/تسليم_صباحي/لقطات/03_قائمة_المراجعة.png filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
reports/تسليم_صباحي/لقطات/04_لوحة_الأدلة.png filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
reports/تسليم_صباحي/لقطات/05_قرار_الباحث.png filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
reports/تسليم_صباحي/لقطات/06_مكتبة_المصادر.png filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
reports/تسليم_صباحي/لقطات/07_اختيار_المصادر.png filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
reports/تسليم_صباحي/لقطات/08_بعد_التنزيل.png filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
reports/تسليم_صباحي/لقطات/09_معاينة_OpenITI.png filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
reports/تسليم_صباحي/لقطات/10_بعد_فهرسة_OpenITI.png filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
reports/تسليم_صباحي/لقطات/11_معاينة_رابط_OpenITI.png filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
reports/تسليم_صباحي/لقطات/12_ترخيص_مصرَّح.png filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
reports/تسليم_صباحي/لقطات/13_IIIF_صور_فقط.png filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
reports/تسليم_صباحي/لقطات/14_IIIF_بنص.png filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
reports/تسليم_صباحي/لقطات/15_سجل_القرارات.png filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
reports/تسليم_صباحي/لقطات/16_بعد_الاسترجاع.png filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
reports/تسليم_صباحي/لقطات/17_كل_الأسطر_قابلة_للفتح.png filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
reports/تسليم_صباحي/لقطات/18_إلغاء_الإثراء.png filter=lfs diff=lfs merge=lfs -text
|
| 25 |
-
reports/تسليم_صباحي/لقطات/19_حالات_المصادر.png filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
samples/logic_router/sample_01_framed_margin.jpg.jpg filter=lfs diff=lfs merge=lfs -text
|
| 27 |
-
samples/logic_router/sample_02_handwritten_logic[[:space:]](1).jpg filter=lfs diff=lfs merge=lfs -text
|
| 28 |
-
samples/logic_router/sample_03_handwritten_logic[[:space:]](2).jpg filter=lfs diff=lfs merge=lfs -text
|
| 29 |
-
samples/logic_router/sample_04_handwritten_logic[[:space:]](3).jpg filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
tmp/book_pages/page_0023.png filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
tmp/book_pages/page_0024.png filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
tmp/book_pages/page_0025.png filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
tmp/book_pages/page_0026.png filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
tmp/book_pages/page_0027.png filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
tmp/book_pages/page_0028.png filter=lfs diff=lfs merge=lfs -text
|
| 36 |
-
tmp/book_pages/page_0029.png filter=lfs diff=lfs merge=lfs -text
|
| 37 |
-
tmp/book_pages/page_0030.png filter=lfs diff=lfs merge=lfs -text
|
| 38 |
-
tmp/book_pages/page_0031.png filter=lfs diff=lfs merge=lfs -text
|
| 39 |
-
tmp/book_pages/page_0032.png filter=lfs diff=lfs merge=lfs -text
|
| 40 |
-
tmp/book_pages/page_0033.png filter=lfs diff=lfs merge=lfs -text
|
| 41 |
-
tmp/book_pages/page_0034.png filter=lfs diff=lfs merge=lfs -text
|
| 42 |
-
tmp/book_pages/page_0035.png filter=lfs diff=lfs merge=lfs -text
|
| 43 |
-
tmp/book_pages/page_0036.png filter=lfs diff=lfs merge=lfs -text
|
| 44 |
-
tmp/book_pages/page_0037.png filter=lfs diff=lfs merge=lfs -text
|
| 45 |
-
tmp/book_pages/page_0038.png filter=lfs diff=lfs merge=lfs -text
|
| 46 |
-
tmp/book_pages/page_0039.png filter=lfs diff=lfs merge=lfs -text
|
| 47 |
-
tmp/book_pages/page_0040.png filter=lfs diff=lfs merge=lfs -text
|
| 48 |
-
tmp/book_pages/page_0041.png filter=lfs diff=lfs merge=lfs -text
|
| 49 |
-
tmp/pdfs/PAPER_v2_page1.png filter=lfs diff=lfs merge=lfs -text
|
| 50 |
-
tmp/pdfs/project_submission_review/page-01.png filter=lfs diff=lfs merge=lfs -text
|
| 51 |
-
tmp/pdfs/project_submission_review/page-02.png filter=lfs diff=lfs merge=lfs -text
|
| 52 |
-
tmp/pdfs/project_submission_review/page-03.png filter=lfs diff=lfs merge=lfs -text
|
| 53 |
-
tmp/pdfs/project_submission_review/page-04.png filter=lfs diff=lfs merge=lfs -text
|
| 54 |
-
tmp/pdfs/project_submission_review/page-05.png filter=lfs diff=lfs merge=lfs -text
|
| 55 |
-
tmp/pdfs/project_submission_review/page-06.png filter=lfs diff=lfs merge=lfs -text
|
| 56 |
-
top_15_easiest_images/BULAC_ARA_MS_1982_182182.jpg filter=lfs diff=lfs merge=lfs -text
|
| 57 |
-
top_15_easiest_images/BULAC_MS_ARA_1926_0077.jpg filter=lfs diff=lfs merge=lfs -text
|
| 58 |
-
top_15_easiest_images/BULAC_MS_ARA_1926_0144.jpg filter=lfs diff=lfs merge=lfs -text
|
| 59 |
-
top_15_easiest_images/BULAC_MS_ARA_1943_180499.jpg filter=lfs diff=lfs merge=lfs -text
|
| 60 |
-
top_15_easiest_images/BULAC_MS_ARA_1944_0006.jpg filter=lfs diff=lfs merge=lfs -text
|
| 61 |
-
top_15_easiest_images/BULAC_MS_ARA_1977_0092.jpg filter=lfs diff=lfs merge=lfs -text
|
| 62 |
-
top_15_easiest_images/BULAC_MS_ARA_1977_0093.jpg filter=lfs diff=lfs merge=lfs -text
|
| 63 |
-
top_15_easiest_images/BULAC_MS_ARA_1977_0136.jpg filter=lfs diff=lfs merge=lfs -text
|
| 64 |
-
top_15_easiest_images/BULAC_MS_ARA_1977_0153.jpg filter=lfs diff=lfs merge=lfs -text
|
| 65 |
-
top_15_easiest_images/BULAC_MS_ARA_1983_177819.jpg filter=lfs diff=lfs merge=lfs -text
|
| 66 |
-
top_15_easiest_images/BULAC_MS_ARA_23_42035.jpg filter=lfs diff=lfs merge=lfs -text
|
| 67 |
-
top_15_easiest_images/BULAC_MS_ARA_417_0009.jpg filter=lfs diff=lfs merge=lfs -text
|
| 68 |
-
top_15_easiest_images/BULAC_MS_ARA_417_0011.jpg filter=lfs diff=lfs merge=lfs -text
|
| 69 |
-
top_15_easiest_images/تفريغ.zip filter=lfs diff=lfs merge=lfs -text
|
| 70 |
-
عرض_قص_آلي/1944_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
|
| 71 |
-
عرض_قص_آلي/1977_0092_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
|
| 72 |
-
عرض_قص_آلي/1977_0136_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
|
| 73 |
-
عرض_قص_آلي/1983_شرقي_سهل_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
|
| 74 |
-
عرض_قص_آلي/417_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
|
| 75 |
-
عرض_قص_آلي/منطق_page16_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
|
| 76 |
-
عرض_قص_آلي/منطق_page17_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
|
| 77 |
-
عرض_منطق_قص_GT/تحرير_القواعد_المنطقية_في_شرح_الرسالة_الشمسية.pdf_page_16.jpg filter=lfs diff=lfs merge=lfs -text
|
| 78 |
-
عرض_منطق_قص_GT/تحرير_القواعد_المنطقية_في_شرح_الرسالة_الشمسية.pdf_page_17.jpg filter=lfs diff=lfs merge=lfs -text
|
|
|
|
| 1 |
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 2 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
PUBLISH_CHECKLIST.md
CHANGED
|
@@ -1,15 +1,15 @@
|
|
| 1 |
# Hugging Face publication checklist
|
| 2 |
|
| 3 |
-
|
| 4 |
|
| 5 |
`https://huggingface.co/factlogic/phoenix-arabic-manuscript-htr`
|
| 6 |
|
| 7 |
-
The
|
| 8 |
|
| 9 |
## Before publishing
|
| 10 |
|
| 11 |
1. Confirm the repository name and owner.
|
| 12 |
-
2.
|
| 13 |
3. Verify `SHA256SUMS` against `model.mlmodel`.
|
| 14 |
4. Confirm that the README has no placeholder comments.
|
| 15 |
5. Confirm that aggregate benchmark JSON contains no line-level ground truth or restricted images.
|
|
@@ -28,4 +28,4 @@ $hf = "$env:LOCALAPPDATA\Programs\Python\Python313\Scripts\hf.exe"
|
|
| 28 |
& $hf upload factlogic/phoenix-arabic-manuscript-htr . . --repo-type model
|
| 29 |
```
|
| 30 |
|
| 31 |
-
Run the upload command from this directory.
|
|
|
|
| 1 |
# Hugging Face publication checklist
|
| 2 |
|
| 3 |
+
Published publicly on 2026-08-17 at:
|
| 4 |
|
| 5 |
`https://huggingface.co/factlogic/phoenix-arabic-manuscript-htr`
|
| 6 |
|
| 7 |
+
The public release uses the conservative CC BY-NC-SA 2.0 license because Muharaf contributed to training. The checks below remain the release audit trail.
|
| 8 |
|
| 9 |
## Before publishing
|
| 10 |
|
| 11 |
1. Confirm the repository name and owner.
|
| 12 |
+
2. Preserve the CC BY-NC-SA 2.0 license and the non-commercial training-data notice unless a new legal review authorizes a change.
|
| 13 |
3. Verify `SHA256SUMS` against `model.mlmodel`.
|
| 14 |
4. Confirm that the README has no placeholder comments.
|
| 15 |
5. Confirm that aggregate benchmark JSON contains no line-level ground truth or restricted images.
|
|
|
|
| 28 |
& $hf upload factlogic/phoenix-arabic-manuscript-htr . . --repo-type model
|
| 29 |
```
|
| 30 |
|
| 31 |
+
Run the upload command from this directory. The repository is already public; do not change the license metadata without reviewing upstream dataset obligations.
|
README.md
CHANGED
|
@@ -1,87 +1,254 @@
|
|
| 1 |
-
|
| 2 |
-
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
| 8 |
-
|
| 9 |
-
|
| 10 |
-
|
| 11 |
-
|
| 12 |
-
|
| 13 |
-
|
| 14 |
-
|
| 15 |
-
|
| 16 |
-
|
| 17 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 18 |
|
| 19 |
-
|
| 20 |
|
| 21 |
-
-
|
| 22 |
-
- قراءة عربية متعددة المجالات بنموذج exp9.
|
| 23 |
-
- قراءات بصرية بديلة مع درجات نسبية غير معروضة كاحتمالات معايرة.
|
| 24 |
-
- نموذج لغوي محلي مشروط بالمجال والثقة، مع إبقاء القراءة الخام.
|
| 25 |
-
- اقتراح LLM خارجي اختياري وموسوم بوصفه استشارياً.
|
| 26 |
-
- مكتبة مصادر محلية قابلة للتوسعة، وامتناع عند غموض الإسناد.
|
| 27 |
-
- قائمة مراجعة، وسجل قرارات الباحث، وتصدير PAGE-XML/TEI.
|
| 28 |
-
- حزمة تدريب تستبعد المخرجات الآلية غير المعتمدة بشرياً افتراضياً.
|
| 29 |
|
| 30 |
-
|
| 31 |
|
| 32 |
-
|
| 33 |
|
| 34 |
-
|
|
|
|
|
|
|
| 35 |
|---|---|---|
|
| 36 |
-
|
|
| 37 |
-
|
|
|
|
|
|
|
|
|
|
|
| 38 |
|
| 39 |
-
|
| 40 |
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
|
| 44 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
|
| 46 |
-
|
| 47 |
-
|
|
|
|
|
|
|
|
|
|
| 48 |
|
| 49 |
-
##
|
| 50 |
|
| 51 |
-
|
| 52 |
|
| 53 |
-
```
|
| 54 |
-
|
| 55 |
-
python -m venv .venv
|
| 56 |
-
.\.venv\Scripts\python.exe -m pip install -r requirements.txt
|
| 57 |
-
Copy-Item .env.example .env
|
| 58 |
-
cd ..
|
| 59 |
-
.\تشغيل_المشروع.bat
|
| 60 |
```
|
| 61 |
|
| 62 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 63 |
|
| 64 |
-
|
| 65 |
-
في Git؛ لا ترفع المفاتيح إلى المستودع.
|
| 66 |
|
| 67 |
-
|
| 68 |
|
| 69 |
-
-
|
| 70 |
-
-
|
| 71 |
-
-
|
| 72 |
-
-
|
| 73 |
-
-
|
| 74 |
-
-
|
|
|
|
| 75 |
|
| 76 |
-
|
| 77 |
|
| 78 |
-
|
| 79 |
-
|
| 80 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 81 |
|
| 82 |
-
##
|
| 83 |
|
| 84 |
-
|
| 85 |
-
محافظ `CC BY-NC-SA 2.0` بسبب أكثر مصادر تدريبه تقييداً، ومستودع نموذج القص
|
| 86 |
-
خاص إلى أن تكتمل مراجعة أصل نموذج الـwarm-start. راجع بطاقات Hugging Face قبل
|
| 87 |
-
أي نشر عام أو استعمال تجاري.
|
|
|
|
| 1 |
+
---
|
| 2 |
+
language:
|
| 3 |
+
- ar
|
| 4 |
+
library_name: kraken
|
| 5 |
+
pipeline_tag: image-to-text
|
| 6 |
+
tags:
|
| 7 |
+
- htr
|
| 8 |
+
- ocr
|
| 9 |
+
- arabic
|
| 10 |
+
- arabic-manuscripts
|
| 11 |
+
- historical-documents
|
| 12 |
+
- kraken
|
| 13 |
+
- ctc
|
| 14 |
+
- human-in-the-loop
|
| 15 |
+
license: cc-by-nc-sa-2.0
|
| 16 |
+
model-index:
|
| 17 |
+
- name: Phoenix — Arabic Manuscript HTR Model
|
| 18 |
+
results:
|
| 19 |
+
- task:
|
| 20 |
+
type: image-to-text
|
| 21 |
+
name: Arabic Handwritten Text Recognition
|
| 22 |
+
dataset:
|
| 23 |
+
name: Agapet sealed manuscripts
|
| 24 |
+
type: agapet-sealed
|
| 25 |
+
metrics:
|
| 26 |
+
- type: cer
|
| 27 |
+
value: 17.86
|
| 28 |
+
name: Raw CER (%)
|
| 29 |
+
- type: wer
|
| 30 |
+
value: 58.80
|
| 31 |
+
name: Raw WER (%)
|
| 32 |
+
- task:
|
| 33 |
+
type: image-to-text
|
| 34 |
+
name: Arabic Handwritten Text Recognition
|
| 35 |
+
dataset:
|
| 36 |
+
name: Omar document-level sealed split
|
| 37 |
+
type: omar-sealed
|
| 38 |
+
metrics:
|
| 39 |
+
- type: cer
|
| 40 |
+
value: 11.84
|
| 41 |
+
name: Raw CER (%)
|
| 42 |
+
- type: wer
|
| 43 |
+
value: 42.92
|
| 44 |
+
name: Raw WER (%)
|
| 45 |
+
- task:
|
| 46 |
+
type: image-to-text
|
| 47 |
+
name: Arabic Handwritten Text Recognition
|
| 48 |
+
dataset:
|
| 49 |
+
name: TariMa sealed manuscript test
|
| 50 |
+
type: tarima-sealed
|
| 51 |
+
metrics:
|
| 52 |
+
- type: cer
|
| 53 |
+
value: 10.72
|
| 54 |
+
name: Raw CER (%)
|
| 55 |
+
- type: wer
|
| 56 |
+
value: 38.89
|
| 57 |
+
name: Raw WER (%)
|
| 58 |
+
---
|
| 59 |
+
|
| 60 |
+
# Phoenix — Arabic Manuscript HTR Model
|
| 61 |
+
|
| 62 |
+
**Phoenix** is a compact Arabic handwritten-text recognizer for Maghrebi manuscripts, historical manuscripts, and archival documents. Its internal checkpoint identifier is `exp9`; that identifier describes this release checkpoint, not the public model name. Phoenix is the recognizer deployed in the **Athar** human-in-the-loop manuscript investigation system.
|
| 63 |
+
|
| 64 |
+
Phoenix uses a deliberately small **CNN + BiLSTM + CTC** architecture: convolutional layers extract visual features, bidirectional LSTMs model the character sequence, and CTC aligns image features to text without requiring character-level segmentation. It has **4,988,946 parameters** and a **19.94 MB** model file. Depending on manuscript domain and evaluation protocol, observed CER is roughly **7–18%**. In one local batch rehearsal it recognized 288 pre-segmented lines in 32 seconds (about 9 lines/s); this timing is hardware- and pipeline-specific, not a universal latency guarantee.
|
| 65 |
+
|
| 66 |
+
The model is not presented as a universal Arabic OCR system. Its strongest evidence is a preregistered comparison with the preceding `exp8` checkpoint on 22,442 sealed lines: it substantially improved two large external domains and showed a small regression on a third domain. Across the two large sealed sets together (22,278 lines), character-weighted CER fell from **19.98% to 14.93%**, a **25.3% relative reduction in character errors**.
|
| 67 |
+
|
| 68 |
+
## ملخص عربي
|
| 69 |
+
|
| 70 |
+
**Phoenix — Arabic Manuscript HTR Model** نموذج صغير للتعرّف على الكتابة العربية اليدوية، وبنيته **CNN + BiLSTM + CTC** بحوالي 5 ملايين معامل وملف حجمه نحو 20 ميغابايت. حقق CER يتراوح تقريباً بين **7% و18%** باختلاف المجال والبروتوكول. معرّف `exp9` اسم داخلي لنقطة الحفظ الحالية، وليس الاسم العام للنموذج. في مجموعتي Agapet وOmar المختومتين معاً (22,278 سطراً) انخفض CER الموزون بالمحارف من 19.98% إلى 14.93% مقارنةً بنقطة الحفظ السابقة، أي خفض نسبي للأخطاء قدره 25.3%. النموذج جزء من منظومة «أثر» التي تعرض القراءات البديلة وشواهد المصادر وقرار الباحث، لكن هذه الميزات ليست مخزنة داخل الأوزان نفسها.
|
| 71 |
+
|
| 72 |
+
النموذج بحثي وغير تجاري وفق الترخيص المحافظ CC BY-NC-SA 2.0 بسبب أحد مصادر التدريب. لا يُستخدم لإنتاج تحقيق علمي نهائي بلا مراجعة بشرية.
|
| 73 |
+
|
| 74 |
+
## Evidence at a glance
|
| 75 |
+
|
| 76 |
+
| Evidence level | What it measures | Status and headline |
|
| 77 |
+
|---|---|---|
|
| 78 |
+
| **Sealed evaluation** | Frozen `exp8` vs current `exp9` checkpoint on data opened once after selection | Strongest release evidence: **17.86% Agapet**, **11.84% Omar**, **10.72% TariMa** raw CER |
|
| 79 |
+
| **Development diagnostics** | Same-protocol comparisons used for diagnosis and model selection | Useful but not independent final evidence: Phoenix CER spans **7.67–12.31%** across four listed guards |
|
| 80 |
+
| **Additional server benchmark** | Phoenix vs Baseer__Nakba VLM aggregates supplied from a server run | **Preliminary**: Phoenix macro CER **9.71% vs 28.67%**, but Baseer wins Omar and exact-line accuracy |
|
| 81 |
+
|
| 82 |
+
## Model summary
|
| 83 |
+
|
| 84 |
+
| Field | Value |
|
| 85 |
+
|---|---|
|
| 86 |
+
| Public model name | Phoenix — Arabic Manuscript HTR Model |
|
| 87 |
+
| Internal checkpoint ID | `exp9` |
|
| 88 |
+
| Framework | Kraken / PyTorch |
|
| 89 |
+
| Parameters | 4,988,946 |
|
| 90 |
+
| Input | One-channel line image; Kraken resizes to model height 120 |
|
| 91 |
+
| Output codec | 81 symbols + CTC blank |
|
| 92 |
+
| Architecture | CNN + BiLSTM + CTC |
|
| 93 |
+
| Main layers | 4 convolutional blocks + 4 bidirectional LSTM layers + linear CTC output |
|
| 94 |
+
| Model file | `model.mlmodel` |
|
| 95 |
+
| SHA-256 | `2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea` |
|
| 96 |
+
| Intended use | Research-assisted transcription of Arabic handwriting with human review |
|
| 97 |
+
|
| 98 |
+
## Sealed evaluation
|
| 99 |
+
|
| 100 |
+
All values below use the same images, references, and greedy decoder for exp8 and exp9. No language model is included in these figures. The sealed sets were opened once after the two arms were frozen; exp9 was not tuned after seeing these results.
|
| 101 |
+
|
| 102 |
+
| Sealed set | Lines | exp8 raw CER | exp9 raw CER | Change | exp8 → exp9 word accuracy |
|
| 103 |
+
|---|---:|---:|---:|---:|---:|
|
| 104 |
+
| Agapet: Sin423 + BnF Arabe 76 | 10,594 | 22.12% | **17.86%** | **−4.26 pp** | 23.91% → **33.23%** |
|
| 105 |
+
| Omar: 11 held-out documents | 11,684 | 17.72% | **11.84%** | **−5.88 pp** | 33.40% → **46.83%** |
|
| 106 |
+
| TariMa manuscript test | 164 | **10.39%** | 10.72% | +0.33 pp | 51.32% → 49.23% |
|
| 107 |
+
|
| 108 |
+
Both Agapet manuscripts improved, and all eleven Omar documents improved. TariMa is the declared exception: exp9 regressed by 0.33 CER percentage points, within the 0.5-point tolerance registered before opening the sealed results.
|
| 109 |
+
|
| 110 |
+
## Preliminary additional benchmark: Phoenix vs Baseer__Nakba
|
| 111 |
+
|
| 112 |
+
In a **preliminary owner-run server comparison**, Phoenix achieved a substantially lower **four-domain macro CER** and **overall WER** than the multi-billion-parameter Baseer__Nakba VLM while using roughly three orders of magnitude fewer parameters. The comparison also exposes an important counter-result: Baseer__Nakba had higher exact-line accuracy and was markedly better on Omar. These aggregate values are reported as supplied; the raw per-line predictions, sample counts, preprocessing manifest, and executable evaluation bundle are not yet included in this repository. This is therefore preliminary supporting evidence, not a sealed or independently reproduced benchmark.
|
| 113 |
+
|
| 114 |
+
| Metric | Phoenix | Baseer__Nakba | Result |
|
| 115 |
+
|---|---:|---:|---|
|
| 116 |
+
| Four-domain unweighted macro CER | **9.71%** | 28.67% | Phoenix: **66.1% relative CER reduction** |
|
| 117 |
+
| Overall WER | **34.37%** | 59.52% | Phoenix: **−25.15 pp**, 42.3% relative reduction |
|
| 118 |
+
| Exact-line accuracy | 13.38% | **26.38%** | Baseer__Nakba higher |
|
| 119 |
+
| Parameters | **4,988,946** | ≈3.75B | Phoenix ≈**752× smaller** |
|
| 120 |
+
| Model file | **19.94 MB** | ≈7.53 GB | Phoenix ≈**378× smaller** |
|
| 121 |
+
|
| 122 |
+
### Raw CER by dataset
|
| 123 |
+
|
| 124 |
+
| Dataset | Phoenix | Baseer__Nakba | Lower CER |
|
| 125 |
+
|---|---:|---:|---|
|
| 126 |
+
| Agapet | **11.13%** | 35.65% | Phoenix (68.8% relative reduction) |
|
| 127 |
+
| Muharaf | **11.93%** | 25.51% | Phoenix (53.2% relative reduction) |
|
| 128 |
+
| Omar | 6.88% | **0.48%** | Baseer__Nakba |
|
| 129 |
+
| RASAM | **8.90%** | 53.06% | Phoenix (83.2% relative reduction) |
|
| 130 |
+
|
| 131 |
+
The macro CER is an **unweighted mean of the four dataset CER values**, so each dataset contributes equally regardless of line or character count. The four displayed Baseer values average to 28.675%; the supplied 28.67% display is retained, while the unrounded recomputation is recorded in `baseer_server_benchmark.json`. The supported conclusion is therefore: **under this server protocol, Phoenix has lower CER on three of four datasets and a much lower macro CER at a fraction of the model size; Baseer__Nakba remains stronger on Omar and exact-line accuracy.**
|
| 132 |
+
|
| 133 |
+
## Same-protocol diagnostic comparison
|
| 134 |
+
|
| 135 |
+
All models below were decoded on the same pre-cropped line images and raw references with Kraken greedy decoding, without an LM or normalization. Each cell is **CER / word accuracy**.
|
| 136 |
+
|
| 137 |
+
| Development guard | Lines | Original Muharaf | exp6 | exp8 | **exp9** |
|
| 138 |
+
|---|---:|---:|---:|---:|---:|
|
| 139 |
+
| Agapet | 991 | 33.15 / 16.62 | 23.82 / 24.71 | 20.08 / 33.58 | **9.47 / 67.40** |
|
| 140 |
+
| Omar | 1,143 | 9.26 / 60.08 | 26.88 / 21.25 | 12.06 / 53.90 | **8.91 / 63.95** |
|
| 141 |
+
| RASAM | 1,789 | 39.01 / 12.06 | 9.04 / 66.78 | 7.95 / 69.90 | **7.67 / 70.46** |
|
| 142 |
+
| Muharaf | 920 | 13.28 / **64.15** | 36.47 / 15.74 | 12.72 / 58.30 | **12.31** / 59.32 |
|
| 143 |
+
| **Four-domain unweighted macro** | **4,843** | 23.68 / 38.23 | 24.05 / 32.12 | 13.20 / 53.92 | **9.59 / 65.28** |
|
| 144 |
|
| 145 |
+
Under this exact diagnostic protocol, exp9 had the lowest CER on **4/4** comparable guards and the highest word accuracy on **3/4**. Its macro CER was 9.59% versus 23.68% for the original Muharaf model, a 59.5% relative reduction in character errors. On the Muharaf guard itself, exp9 had lower CER (12.31% vs 13.28%) but lower word accuracy (59.32% vs 64.15%); both metrics are reported to avoid hiding the trade-off.
|
| 146 |
|
| 147 |
+
The 369-line TariMa manuscript guard is excluded from this macro because it was independent for exp9 but not for exp6/exp8. Its scores are retained in `BENCHMARKS.md` as a non-ranked diagnostic.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 148 |
|
| 149 |
+
This table is a development diagnostic, not an independent final test: some of the guard sets participated in exp9 selection. Its purpose is to compare `muharaf_rec_best`, exp6, exp8, and exp9 under one raw-reference protocol. It must not be mixed with published Muharaf numbers obtained from a different split, codec, or normalization policy.
|
| 150 |
|
| 151 |
+
## Training data
|
| 152 |
|
| 153 |
+
exp9 used document-aware training/replay from five sources:
|
| 154 |
+
|
| 155 |
+
| Dataset | Role | Recorded license |
|
| 156 |
|---|---|---|
|
| 157 |
+
| Muharaf public line images | Archival handwriting | CC BY-NC-SA 2.0 |
|
| 158 |
+
| RASAM | Maghrebi manuscripts | Apache-2.0 in the local dataset repository |
|
| 159 |
+
| TariMa | Maghrebi/historical manuscripts | Apache-2.0 |
|
| 160 |
+
| Agapet SA-418 | 13th-century historical manuscript | CC BY 4.0 |
|
| 161 |
+
| Omar Al-Saleh manuscripts | Diverse archival documents | CC BY 4.0; access-gated at download time |
|
| 162 |
|
| 163 |
+
Training and guard splits were separated at document or manuscript level where the source allowed it. The Agapet sealed manuscripts and the eleven Omar sealed documents did not enter training.
|
| 164 |
|
| 165 |
+
Because Muharaf is CC BY-NC-SA 2.0, this release uses the same non-commercial ShareAlike license as the conservative publication choice. Users are responsible for checking whether their intended use and jurisdiction are compatible with every upstream dataset license.
|
| 166 |
+
|
| 167 |
+
## Intended use
|
| 168 |
+
|
| 169 |
+
- Research and non-commercial transcription assistance for Arabic manuscripts and archival documents.
|
| 170 |
+
- Producing initial transcriptions for expert review.
|
| 171 |
+
- Generating multiple CTC candidates for a human-in-the-loop workflow.
|
| 172 |
+
- Use with PAGE-XML line polygons when layout segmentation is supplied externally.
|
| 173 |
+
|
| 174 |
+
## Out-of-scope use
|
| 175 |
|
| 176 |
+
- Fully automatic scholarly editions without expert review.
|
| 177 |
+
- Claims of universal accuracy across all Arabic scripts, periods, or image conditions.
|
| 178 |
+
- Automatic attribution of text to a unique source without a separate retrieval/attribution layer.
|
| 179 |
+
- Commercial use without a separate legal review of the training-data obligations.
|
| 180 |
+
- Treating decoder scores as calibrated probabilities of correctness.
|
| 181 |
|
| 182 |
+
## Inference
|
| 183 |
|
| 184 |
+
Install a Kraken version compatible with PyTorch 2.4, then use the model as a Kraken recognition model. A typical page command is:
|
| 185 |
|
| 186 |
+
```bash
|
| 187 |
+
kraken -i page.jpg output.txt segment ocr -m model.mlmodel
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 188 |
```
|
| 189 |
|
| 190 |
+
For complex pages, the recommended Athar workflow supplies PAGE-XML polygons and runs recognition on the human-defined line regions. This avoids mixing segmentation failure with recognition error.
|
| 191 |
+
|
| 192 |
+
## Language-model policy in Athar
|
| 193 |
+
|
| 194 |
+
The published CER/WER values are greedy recognition values. The complete Athar system can additionally:
|
| 195 |
+
|
| 196 |
+
- preserve the raw visual reading for every line;
|
| 197 |
+
- generate up to eight visual candidates;
|
| 198 |
+
- rerank candidates with a small local character n-gram model;
|
| 199 |
+
- apply the general/Maghrebi LM automatically only in a conservative confidence band;
|
| 200 |
+
- show an archival LM as advisory evidence instead of silently replacing the text;
|
| 201 |
+
- protect numbers and punctuation from destructive LM changes.
|
| 202 |
+
|
| 203 |
+
On a 1,789-line RASAM development guard, the deployed conservative general-LM policy produced only a small change (about −0.003 CER points and −0.141 WER points). The large exp9 gains therefore come from the visual recognizer, not from post-hoc language correction.
|
| 204 |
|
| 205 |
+
## Athar system capabilities
|
|
|
|
| 206 |
|
| 207 |
+
The `.mlmodel` file is one component of the larger architecture. The application adds:
|
| 208 |
|
| 209 |
+
- visual and language-ranked alternatives;
|
| 210 |
+
- source retrieval with unique/ambiguous attribution states;
|
| 211 |
+
- a review queue with reasons;
|
| 212 |
+
- optional external-LLM suggestions labelled as advisory;
|
| 213 |
+
- auditable human accept/edit/reject decisions;
|
| 214 |
+
- PAGE-XML and TEI export;
|
| 215 |
+
- a training package that excludes unreviewed automatic output by default.
|
| 216 |
|
| 217 |
+
These are system capabilities, not properties encoded inside the model weights.
|
| 218 |
|
| 219 |
+
## Limitations
|
| 220 |
+
|
| 221 |
+
- Raw CER remains 17.86% on the difficult Agapet sealed set; human review is still necessary.
|
| 222 |
+
- TariMa sealed performance is 0.33 CER points worse than exp8.
|
| 223 |
+
- The model is optimized for Arabic handwriting, not modern printed OCR.
|
| 224 |
+
- Layout/segmentation quality can dominate full-page performance.
|
| 225 |
+
- The 81-symbol codec does not cover every possible Arabic-script or Latin character.
|
| 226 |
+
- Confidence values and beam scores are not globally calibrated probabilities.
|
| 227 |
+
- The same-protocol Muharaf comparison in this repository is a development comparison, not a new sealed benchmark.
|
| 228 |
+
|
| 229 |
+
## Reproducibility
|
| 230 |
+
|
| 231 |
+
- Model SHA-256: `2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea`
|
| 232 |
+
- Python used in the verified local environment: 3.10.11
|
| 233 |
+
- PyTorch: 2.4.1+cu121
|
| 234 |
+
- Evaluation decoder: greedy, without LM
|
| 235 |
+
- Raw and Arabic-normalized metrics are stored separately; the model card reports raw metrics unless explicitly labelled otherwise.
|
| 236 |
+
|
| 237 |
+
See `BENCHMARKS.md`, `benchmark_summary.json`, `baseer_server_benchmark.json`, `DATA_AND_LICENSES.md`, and `metadata.json` in this repository for details.
|
| 238 |
+
|
| 239 |
+
## Citation
|
| 240 |
+
|
| 241 |
+
Until a final paper record is available, cite the project and this model version as:
|
| 242 |
+
|
| 243 |
+
```bibtex
|
| 244 |
+
@misc{athar_htr_exp9_2026,
|
| 245 |
+
title = {Phoenix: Arabic Manuscript HTR Model},
|
| 246 |
+
author = {Athar Project Team},
|
| 247 |
+
year = {2026},
|
| 248 |
+
note = {Internal checkpoint exp9; Kraken CNN-BiLSTM-CTC model; SHA-256 2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea}
|
| 249 |
+
}
|
| 250 |
+
```
|
| 251 |
|
| 252 |
+
## Ethical and scholarly note
|
| 253 |
|
| 254 |
+
The output is a research aid, not an authoritative edition. Manuscript transcription and source attribution require domain expertise. Preserve the image, the raw reading, the evidence trail, and the researcher's final decision.
|
|
|
|
|
|
|
|
|
SHA256SUMS
CHANGED
|
@@ -1,4 +1,4 @@
|
|
| 1 |
2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea model.mlmodel
|
| 2 |
7184ae01c9f4a248d80ddeb84d852a8c21b8d7b213ca7619433edf4b18bf2fe7 benchmark_summary.json
|
| 3 |
-
|
| 4 |
1cc91f05e0ff7e5594b13029a0b82d0fa2b78de3dff1a16160d99ed61441f039 baseer_server_benchmark.json
|
|
|
|
| 1 |
2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea model.mlmodel
|
| 2 |
7184ae01c9f4a248d80ddeb84d852a8c21b8d7b213ca7619433edf4b18bf2fe7 benchmark_summary.json
|
| 3 |
+
d943903a68dd40859b42a66f244d8441b022a8b051a8b1d0716781b7ef9919e4 metadata.json
|
| 4 |
1cc91f05e0ff7e5594b13029a0b82d0fa2b78de3dff1a16160d99ed61441f039 baseer_server_benchmark.json
|
metadata.json
CHANGED
|
@@ -12,7 +12,7 @@
|
|
| 12 |
"python": "3.10.11",
|
| 13 |
"pytorch": "2.4.1+cu121",
|
| 14 |
"created_date": "2026-08-11",
|
| 15 |
-
"release_status": "
|
| 16 |
"license": "CC BY-NC-SA 2.0 (conservative choice due to upstream Muharaf training data)",
|
| 17 |
"sealed_release_evidence": {
|
| 18 |
"lines": 22442,
|
|
|
|
| 12 |
"python": "3.10.11",
|
| 13 |
"pytorch": "2.4.1+cu121",
|
| 14 |
"created_date": "2026-08-11",
|
| 15 |
+
"release_status": "published_public_2026-08-17",
|
| 16 |
"license": "CC BY-NC-SA 2.0 (conservative choice due to upstream Muharaf training data)",
|
| 17 |
"sealed_release_evidence": {
|
| 18 |
"lines": 22442,
|