factlogic commited on
Commit
30fe47d
·
verified ·
1 Parent(s): 84622db

Record public Phoenix release status

Browse files
Files changed (5) hide show
  1. .gitattributes +0 -76
  2. PUBLISH_CHECKLIST.md +4 -4
  3. README.md +233 -66
  4. SHA256SUMS +1 -1
  5. metadata.json +1 -1
.gitattributes CHANGED
@@ -1,78 +1,2 @@
1
  *.mlmodel filter=lfs diff=lfs merge=lfs -text
2
 
3
- presentation_dashboard/assets/before_after/manuscript.png filter=lfs diff=lfs merge=lfs -text
4
- presentation_dashboard/assets/samples/clear/manuscript.jpg filter=lfs diff=lfs merge=lfs -text
5
- presentation_dashboard/assets/samples/logic/manuscript.png filter=lfs diff=lfs merge=lfs -text
6
- presentation_dashboard/assets/samples/medium/manuscript.jpg filter=lfs diff=lfs merge=lfs -text
7
- reports/تسليم_صباحي/لقطات/01_تسجيل_الدخول.png filter=lfs diff=lfs merge=lfs -text
8
- reports/تسليم_صباحي/لقطات/02_التفريغ.png filter=lfs diff=lfs merge=lfs -text
9
- reports/تسليم_صباحي/لقطات/03_قائمة_المراجعة.png filter=lfs diff=lfs merge=lfs -text
10
- reports/تسليم_صباحي/لقطات/04_لوحة_الأدلة.png filter=lfs diff=lfs merge=lfs -text
11
- reports/تسليم_صباحي/لقطات/05_قرار_الباحث.png filter=lfs diff=lfs merge=lfs -text
12
- reports/تسليم_صباحي/لقطات/06_مكتبة_المصادر.png filter=lfs diff=lfs merge=lfs -text
13
- reports/تسليم_صباحي/لقطات/07_اختيار_المصادر.png filter=lfs diff=lfs merge=lfs -text
14
- reports/تسليم_صباحي/لقطات/08_بعد_التنزيل.png filter=lfs diff=lfs merge=lfs -text
15
- reports/تسليم_صباحي/لقطات/09_معاينة_OpenITI.png filter=lfs diff=lfs merge=lfs -text
16
- reports/تسليم_صباحي/لقطات/10_بعد_فهرسة_OpenITI.png filter=lfs diff=lfs merge=lfs -text
17
- reports/تسليم_صباحي/لقطات/11_معاينة_رابط_OpenITI.png filter=lfs diff=lfs merge=lfs -text
18
- reports/تسليم_صباحي/لقطات/12_ترخيص_مصرَّح.png filter=lfs diff=lfs merge=lfs -text
19
- reports/تسليم_صباحي/لقطات/13_IIIF_صور_فقط.png filter=lfs diff=lfs merge=lfs -text
20
- reports/تسليم_صباحي/لقطات/14_IIIF_بنص.png filter=lfs diff=lfs merge=lfs -text
21
- reports/تسليم_صباحي/لقطات/15_سجل_القرارات.png filter=lfs diff=lfs merge=lfs -text
22
- reports/تسليم_صباحي/لقطات/16_بعد_الاسترجاع.png filter=lfs diff=lfs merge=lfs -text
23
- reports/تسليم_صباحي/لقطات/17_كل_الأسطر_قابلة_للفتح.png filter=lfs diff=lfs merge=lfs -text
24
- reports/تسليم_صباحي/لقطات/18_إلغاء_الإثراء.png filter=lfs diff=lfs merge=lfs -text
25
- reports/تسليم_صباحي/لقطات/19_حالات_المصادر.png filter=lfs diff=lfs merge=lfs -text
26
- samples/logic_router/sample_01_framed_margin.jpg.jpg filter=lfs diff=lfs merge=lfs -text
27
- samples/logic_router/sample_02_handwritten_logic[[:space:]](1).jpg filter=lfs diff=lfs merge=lfs -text
28
- samples/logic_router/sample_03_handwritten_logic[[:space:]](2).jpg filter=lfs diff=lfs merge=lfs -text
29
- samples/logic_router/sample_04_handwritten_logic[[:space:]](3).jpg filter=lfs diff=lfs merge=lfs -text
30
- tmp/book_pages/page_0023.png filter=lfs diff=lfs merge=lfs -text
31
- tmp/book_pages/page_0024.png filter=lfs diff=lfs merge=lfs -text
32
- tmp/book_pages/page_0025.png filter=lfs diff=lfs merge=lfs -text
33
- tmp/book_pages/page_0026.png filter=lfs diff=lfs merge=lfs -text
34
- tmp/book_pages/page_0027.png filter=lfs diff=lfs merge=lfs -text
35
- tmp/book_pages/page_0028.png filter=lfs diff=lfs merge=lfs -text
36
- tmp/book_pages/page_0029.png filter=lfs diff=lfs merge=lfs -text
37
- tmp/book_pages/page_0030.png filter=lfs diff=lfs merge=lfs -text
38
- tmp/book_pages/page_0031.png filter=lfs diff=lfs merge=lfs -text
39
- tmp/book_pages/page_0032.png filter=lfs diff=lfs merge=lfs -text
40
- tmp/book_pages/page_0033.png filter=lfs diff=lfs merge=lfs -text
41
- tmp/book_pages/page_0034.png filter=lfs diff=lfs merge=lfs -text
42
- tmp/book_pages/page_0035.png filter=lfs diff=lfs merge=lfs -text
43
- tmp/book_pages/page_0036.png filter=lfs diff=lfs merge=lfs -text
44
- tmp/book_pages/page_0037.png filter=lfs diff=lfs merge=lfs -text
45
- tmp/book_pages/page_0038.png filter=lfs diff=lfs merge=lfs -text
46
- tmp/book_pages/page_0039.png filter=lfs diff=lfs merge=lfs -text
47
- tmp/book_pages/page_0040.png filter=lfs diff=lfs merge=lfs -text
48
- tmp/book_pages/page_0041.png filter=lfs diff=lfs merge=lfs -text
49
- tmp/pdfs/PAPER_v2_page1.png filter=lfs diff=lfs merge=lfs -text
50
- tmp/pdfs/project_submission_review/page-01.png filter=lfs diff=lfs merge=lfs -text
51
- tmp/pdfs/project_submission_review/page-02.png filter=lfs diff=lfs merge=lfs -text
52
- tmp/pdfs/project_submission_review/page-03.png filter=lfs diff=lfs merge=lfs -text
53
- tmp/pdfs/project_submission_review/page-04.png filter=lfs diff=lfs merge=lfs -text
54
- tmp/pdfs/project_submission_review/page-05.png filter=lfs diff=lfs merge=lfs -text
55
- tmp/pdfs/project_submission_review/page-06.png filter=lfs diff=lfs merge=lfs -text
56
- top_15_easiest_images/BULAC_ARA_MS_1982_182182.jpg filter=lfs diff=lfs merge=lfs -text
57
- top_15_easiest_images/BULAC_MS_ARA_1926_0077.jpg filter=lfs diff=lfs merge=lfs -text
58
- top_15_easiest_images/BULAC_MS_ARA_1926_0144.jpg filter=lfs diff=lfs merge=lfs -text
59
- top_15_easiest_images/BULAC_MS_ARA_1943_180499.jpg filter=lfs diff=lfs merge=lfs -text
60
- top_15_easiest_images/BULAC_MS_ARA_1944_0006.jpg filter=lfs diff=lfs merge=lfs -text
61
- top_15_easiest_images/BULAC_MS_ARA_1977_0092.jpg filter=lfs diff=lfs merge=lfs -text
62
- top_15_easiest_images/BULAC_MS_ARA_1977_0093.jpg filter=lfs diff=lfs merge=lfs -text
63
- top_15_easiest_images/BULAC_MS_ARA_1977_0136.jpg filter=lfs diff=lfs merge=lfs -text
64
- top_15_easiest_images/BULAC_MS_ARA_1977_0153.jpg filter=lfs diff=lfs merge=lfs -text
65
- top_15_easiest_images/BULAC_MS_ARA_1983_177819.jpg filter=lfs diff=lfs merge=lfs -text
66
- top_15_easiest_images/BULAC_MS_ARA_23_42035.jpg filter=lfs diff=lfs merge=lfs -text
67
- top_15_easiest_images/BULAC_MS_ARA_417_0009.jpg filter=lfs diff=lfs merge=lfs -text
68
- top_15_easiest_images/BULAC_MS_ARA_417_0011.jpg filter=lfs diff=lfs merge=lfs -text
69
- top_15_easiest_images/تفريغ.zip filter=lfs diff=lfs merge=lfs -text
70
- عرض_قص_آلي/1944_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
71
- عرض_قص_آلي/1977_0092_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
72
- عرض_قص_آلي/1977_0136_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
73
- عرض_قص_آلي/1983_شرقي_سهل_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
74
- عرض_قص_آلي/417_مغربي_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
75
- عرض_قص_آلي/منطق_page16_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
76
- عرض_قص_آلي/منطق_page17_قص_آلي.jpg filter=lfs diff=lfs merge=lfs -text
77
- عرض_منطق_قص_GT/تحرير_القواعد_المنطقية_في_شرح_الرسالة_الشمسية.pdf_page_16.jpg filter=lfs diff=lfs merge=lfs -text
78
- عرض_منطق_قص_GT/تحرير_القواعد_المنطقية_في_شرح_الرسالة_الشمسية.pdf_page_17.jpg filter=lfs diff=lfs merge=lfs -text
 
1
  *.mlmodel filter=lfs diff=lfs merge=lfs -text
2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
PUBLISH_CHECKLIST.md CHANGED
@@ -1,15 +1,15 @@
1
  # Hugging Face publication checklist
2
 
3
- Uploaded privately on 2026-08-17 to:
4
 
5
  `https://huggingface.co/factlogic/phoenix-arabic-manuscript-htr`
6
 
7
- The repository remains private pending the license review below.
8
 
9
  ## Before publishing
10
 
11
  1. Confirm the repository name and owner.
12
- 2. Keep the repository private until the training-data/license statement has been reviewed.
13
  3. Verify `SHA256SUMS` against `model.mlmodel`.
14
  4. Confirm that the README has no placeholder comments.
15
  5. Confirm that aggregate benchmark JSON contains no line-level ground truth or restricted images.
@@ -28,4 +28,4 @@ $hf = "$env:LOCALAPPDATA\Programs\Python\Python313\Scripts\hf.exe"
28
  & $hf upload factlogic/phoenix-arabic-manuscript-htr . . --repo-type model
29
  ```
30
 
31
- Run the upload command from this directory. Remove `--private` only after the license decision is final.
 
1
  # Hugging Face publication checklist
2
 
3
+ Published publicly on 2026-08-17 at:
4
 
5
  `https://huggingface.co/factlogic/phoenix-arabic-manuscript-htr`
6
 
7
+ The public release uses the conservative CC BY-NC-SA 2.0 license because Muharaf contributed to training. The checks below remain the release audit trail.
8
 
9
  ## Before publishing
10
 
11
  1. Confirm the repository name and owner.
12
+ 2. Preserve the CC BY-NC-SA 2.0 license and the non-commercial training-data notice unless a new legal review authorizes a change.
13
  3. Verify `SHA256SUMS` against `model.mlmodel`.
14
  4. Confirm that the README has no placeholder comments.
15
  5. Confirm that aggregate benchmark JSON contains no line-level ground truth or restricted images.
 
28
  & $hf upload factlogic/phoenix-arabic-manuscript-htr . . --repo-type model
29
  ```
30
 
31
+ Run the upload command from this directory. The repository is already public; do not change the license metadata without reviewing upstream dataset obligations.
README.md CHANGED
@@ -1,87 +1,254 @@
1
- # أثر — Athar / Phoenix AI
2
-
3
- منصة بحثية مفتوحة البنية لتفريغ المخطوطات العربية وتحقيقها بمشاركة الباحث.
4
- تجمع بين قص الأسطر، والتعرّف البصري، والقراءات البديلة، والنموذج اللغوي
5
- المحلي، والاسترجاع من مكتبة المصادر (RAC)، وقرار الإنسان القابل للتدقيق.
6
-
7
- ## المعمارية
8
-
9
- ```text
10
- صورة صفحة
11
- Athar Segmentation v4 (مضلعات الأسطر وbaselines)
12
- Athar HTR exp9 (مصفوفة CTC وقراءة greedy)
13
- بدائل بصرية + ترتيب لغوي محلي اختياري
14
- شواهد من مكتبة المصادر مع إسناد unique/ambiguous
15
- قبول/تحرير/رفض الباحث
16
- → PAGE-XML أو TEI أو حزمة تدريب بشرية المراجعة
17
- ```
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
18
 
19
- ## أهم الميزات
20
 
21
- - رفع صفحة كاملة أو استعمال PAGE-XML بقص بشري.
22
- - قراءة عربية متعددة المجالات بنموذج exp9.
23
- - قراءات بصرية بديلة مع درجات نسبية غير معروضة كاحتمالات معايرة.
24
- - نموذج لغوي محلي مشروط بالمجال والثقة، مع إبقاء القراءة الخام.
25
- - اقتراح LLM خارجي اختياري وموسوم بوصفه استشارياً.
26
- - مكتبة مصادر محلية قابلة للتوسعة، وامتناع عند غموض الإسناد.
27
- - قائمة مراجعة، وسجل قرارات الباحث، وتصدير PAGE-XML/TEI.
28
- - حزمة تدريب تستبعد المخرجات الآلية غير المعتمدة بشرياً افتراضياً.
29
 
30
- ## النماذج
31
 
32
- الأوزان لا تدخل GitHub. نُشرت نسخة خاصة ومتحققة البصمة على Hugging Face:
33
 
34
- | الوظيفة | المستودع | SHA-256 |
 
 
35
  |---|---|---|
36
- | Phoenix HTR (checkpoint: `exp9`) | `factlogic/phoenix-arabic-manuscript-htr` | `2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea` |
37
- | القص v4 | `factlogic/athar-segmentation-v4` | `dc07cdeefe17598535ea3f5b187f7ddd2463baed909138c9ffd274c88b0e9a24` |
 
 
 
38
 
39
- بعد تسجيل الدخول إلى Hugging Face شغّل:
40
 
41
- ```powershell
42
- hf auth login
43
- .\scripts\download_hf_models.ps1
44
- ```
 
 
 
 
 
 
45
 
46
- النماذج اللغوية والنموذجان المتخصصان في المنطق محفوظة محلياً في `نماذج/`
47
- لكنها ليست جزءاً من مستودعي Hugging Face أعلاه.
 
 
 
48
 
49
- ## التشغيل على Windows
50
 
51
- يتطلب Python 3.10 وبيئة متوافقة مع Kraken/PyTorch.
52
 
53
- ```powershell
54
- cd backend
55
- python -m venv .venv
56
- .\.venv\Scripts\python.exe -m pip install -r requirements.txt
57
- Copy-Item .env.example .env
58
- cd ..
59
- .\تشغيل_المشروع.bat
60
  ```
61
 
62
- ثم افتح `http://127.0.0.1:8001/app/`.
 
 
 
 
 
 
 
 
 
 
 
 
 
63
 
64
- يمكن وضع مفتاح OpenRouter أو مزود LLM في `backend/.env`. الملف محلي ومُتجاهل
65
- في Git؛ لا ترفع المفاتيح إلى المستودع.
66
 
67
- ## هيكل المشروع
68
 
69
- - `backend/app/`: FastAPI وخدمات OCR/LM/RAC والتصدير.
70
- - `frontend/`: واجهة التحقيق والمراجعة.
71
- - `digital_texts/`: مكتبة الاسترجاع المحلية وسجلات مصادرها.
72
- - `scripts/`: القياس، التدريب، التدقيق، وأدوات التشغيل.
73
- - `reports/`: أدلة القياس والادعاءات المنهجية.
74
- - `huggingface/`: بطاقات النماذج وبيانات نشرها، من دون أوزان داخل Git.
 
75
 
76
- ## البيانات وإعادة الإنتاج
77
 
78
- بيانات التدريب والتقييم الكبيرة محفوظة خارج Git ولا تُرفع مع الكود. افصل
79
- التقسيمات على مستوى المخطوطة/الوثيقة، ولا تضبط المعاملات على مجموعة مختومة.
80
- درجات CTC والـbeam نسبية وليست احتمالات ثقة معايرة.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
81
 
82
- ## الترخيص
83
 
84
- الكود ووزن النموذج ليسا بالضرورة تحت الترخيص نفسه. نموذج exp9 منشور بترخيص
85
- محافظ `CC BY-NC-SA 2.0` بسبب أكثر مصادر تدريبه تقييداً، ومستودع نموذج القص
86
- خاص إلى أن تكتمل مراجعة أصل نموذج الـwarm-start. راجع بطاقات Hugging Face قبل
87
- أي نشر عام أو استعمال تجاري.
 
1
+ ---
2
+ language:
3
+ - ar
4
+ library_name: kraken
5
+ pipeline_tag: image-to-text
6
+ tags:
7
+ - htr
8
+ - ocr
9
+ - arabic
10
+ - arabic-manuscripts
11
+ - historical-documents
12
+ - kraken
13
+ - ctc
14
+ - human-in-the-loop
15
+ license: cc-by-nc-sa-2.0
16
+ model-index:
17
+ - name: Phoenix — Arabic Manuscript HTR Model
18
+ results:
19
+ - task:
20
+ type: image-to-text
21
+ name: Arabic Handwritten Text Recognition
22
+ dataset:
23
+ name: Agapet sealed manuscripts
24
+ type: agapet-sealed
25
+ metrics:
26
+ - type: cer
27
+ value: 17.86
28
+ name: Raw CER (%)
29
+ - type: wer
30
+ value: 58.80
31
+ name: Raw WER (%)
32
+ - task:
33
+ type: image-to-text
34
+ name: Arabic Handwritten Text Recognition
35
+ dataset:
36
+ name: Omar document-level sealed split
37
+ type: omar-sealed
38
+ metrics:
39
+ - type: cer
40
+ value: 11.84
41
+ name: Raw CER (%)
42
+ - type: wer
43
+ value: 42.92
44
+ name: Raw WER (%)
45
+ - task:
46
+ type: image-to-text
47
+ name: Arabic Handwritten Text Recognition
48
+ dataset:
49
+ name: TariMa sealed manuscript test
50
+ type: tarima-sealed
51
+ metrics:
52
+ - type: cer
53
+ value: 10.72
54
+ name: Raw CER (%)
55
+ - type: wer
56
+ value: 38.89
57
+ name: Raw WER (%)
58
+ ---
59
+
60
+ # Phoenix — Arabic Manuscript HTR Model
61
+
62
+ **Phoenix** is a compact Arabic handwritten-text recognizer for Maghrebi manuscripts, historical manuscripts, and archival documents. Its internal checkpoint identifier is `exp9`; that identifier describes this release checkpoint, not the public model name. Phoenix is the recognizer deployed in the **Athar** human-in-the-loop manuscript investigation system.
63
+
64
+ Phoenix uses a deliberately small **CNN + BiLSTM + CTC** architecture: convolutional layers extract visual features, bidirectional LSTMs model the character sequence, and CTC aligns image features to text without requiring character-level segmentation. It has **4,988,946 parameters** and a **19.94 MB** model file. Depending on manuscript domain and evaluation protocol, observed CER is roughly **7–18%**. In one local batch rehearsal it recognized 288 pre-segmented lines in 32 seconds (about 9 lines/s); this timing is hardware- and pipeline-specific, not a universal latency guarantee.
65
+
66
+ The model is not presented as a universal Arabic OCR system. Its strongest evidence is a preregistered comparison with the preceding `exp8` checkpoint on 22,442 sealed lines: it substantially improved two large external domains and showed a small regression on a third domain. Across the two large sealed sets together (22,278 lines), character-weighted CER fell from **19.98% to 14.93%**, a **25.3% relative reduction in character errors**.
67
+
68
+ ## ملخص عربي
69
+
70
+ **Phoenix — Arabic Manuscript HTR Model** نموذج صغير للتعرّف على الكتابة العربية اليدوية، وبنيته **CNN + BiLSTM + CTC** بحوالي 5 ملايين معامل وملف حجمه نحو 20 ميغابايت. حقق CER يتراوح تقريباً بين **7% و18%** باختلاف المجال والبروتوكول. معرّف `exp9` اسم داخلي لنقطة الحفظ الحالية، وليس الاسم العام للنموذج. في مجموعتي Agapet وOmar المختومتين معاً (22,278 سطراً) انخفض CER الموزون بالمحارف من 19.98% إلى 14.93% مقارنةً بنقطة الحفظ السابقة، أي خفض نسبي للأخطاء قدره 25.3%. النموذج جزء من منظومة «أثر» التي تعرض القراءات البديلة وشواهد المصادر وقرار الباحث، لكن هذه الميزات ليست مخزنة داخل الأوزان نفسها.
71
+
72
+ النموذج بحثي وغير تجاري وفق الترخيص المحافظ CC BY-NC-SA 2.0 بسبب أحد مصادر التدريب. لا يُستخدم لإنتاج تحقيق علمي نهائي بلا مراجعة بشرية.
73
+
74
+ ## Evidence at a glance
75
+
76
+ | Evidence level | What it measures | Status and headline |
77
+ |---|---|---|
78
+ | **Sealed evaluation** | Frozen `exp8` vs current `exp9` checkpoint on data opened once after selection | Strongest release evidence: **17.86% Agapet**, **11.84% Omar**, **10.72% TariMa** raw CER |
79
+ | **Development diagnostics** | Same-protocol comparisons used for diagnosis and model selection | Useful but not independent final evidence: Phoenix CER spans **7.67–12.31%** across four listed guards |
80
+ | **Additional server benchmark** | Phoenix vs Baseer__Nakba VLM aggregates supplied from a server run | **Preliminary**: Phoenix macro CER **9.71% vs 28.67%**, but Baseer wins Omar and exact-line accuracy |
81
+
82
+ ## Model summary
83
+
84
+ | Field | Value |
85
+ |---|---|
86
+ | Public model name | Phoenix — Arabic Manuscript HTR Model |
87
+ | Internal checkpoint ID | `exp9` |
88
+ | Framework | Kraken / PyTorch |
89
+ | Parameters | 4,988,946 |
90
+ | Input | One-channel line image; Kraken resizes to model height 120 |
91
+ | Output codec | 81 symbols + CTC blank |
92
+ | Architecture | CNN + BiLSTM + CTC |
93
+ | Main layers | 4 convolutional blocks + 4 bidirectional LSTM layers + linear CTC output |
94
+ | Model file | `model.mlmodel` |
95
+ | SHA-256 | `2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea` |
96
+ | Intended use | Research-assisted transcription of Arabic handwriting with human review |
97
+
98
+ ## Sealed evaluation
99
+
100
+ All values below use the same images, references, and greedy decoder for exp8 and exp9. No language model is included in these figures. The sealed sets were opened once after the two arms were frozen; exp9 was not tuned after seeing these results.
101
+
102
+ | Sealed set | Lines | exp8 raw CER | exp9 raw CER | Change | exp8 → exp9 word accuracy |
103
+ |---|---:|---:|---:|---:|---:|
104
+ | Agapet: Sin423 + BnF Arabe 76 | 10,594 | 22.12% | **17.86%** | **−4.26 pp** | 23.91% → **33.23%** |
105
+ | Omar: 11 held-out documents | 11,684 | 17.72% | **11.84%** | **−5.88 pp** | 33.40% → **46.83%** |
106
+ | TariMa manuscript test | 164 | **10.39%** | 10.72% | +0.33 pp | 51.32% → 49.23% |
107
+
108
+ Both Agapet manuscripts improved, and all eleven Omar documents improved. TariMa is the declared exception: exp9 regressed by 0.33 CER percentage points, within the 0.5-point tolerance registered before opening the sealed results.
109
+
110
+ ## Preliminary additional benchmark: Phoenix vs Baseer__Nakba
111
+
112
+ In a **preliminary owner-run server comparison**, Phoenix achieved a substantially lower **four-domain macro CER** and **overall WER** than the multi-billion-parameter Baseer__Nakba VLM while using roughly three orders of magnitude fewer parameters. The comparison also exposes an important counter-result: Baseer__Nakba had higher exact-line accuracy and was markedly better on Omar. These aggregate values are reported as supplied; the raw per-line predictions, sample counts, preprocessing manifest, and executable evaluation bundle are not yet included in this repository. This is therefore preliminary supporting evidence, not a sealed or independently reproduced benchmark.
113
+
114
+ | Metric | Phoenix | Baseer__Nakba | Result |
115
+ |---|---:|---:|---|
116
+ | Four-domain unweighted macro CER | **9.71%** | 28.67% | Phoenix: **66.1% relative CER reduction** |
117
+ | Overall WER | **34.37%** | 59.52% | Phoenix: **−25.15 pp**, 42.3% relative reduction |
118
+ | Exact-line accuracy | 13.38% | **26.38%** | Baseer__Nakba higher |
119
+ | Parameters | **4,988,946** | ≈3.75B | Phoenix ≈**752× smaller** |
120
+ | Model file | **19.94 MB** | ≈7.53 GB | Phoenix ≈**378× smaller** |
121
+
122
+ ### Raw CER by dataset
123
+
124
+ | Dataset | Phoenix | Baseer__Nakba | Lower CER |
125
+ |---|---:|---:|---|
126
+ | Agapet | **11.13%** | 35.65% | Phoenix (68.8% relative reduction) |
127
+ | Muharaf | **11.93%** | 25.51% | Phoenix (53.2% relative reduction) |
128
+ | Omar | 6.88% | **0.48%** | Baseer__Nakba |
129
+ | RASAM | **8.90%** | 53.06% | Phoenix (83.2% relative reduction) |
130
+
131
+ The macro CER is an **unweighted mean of the four dataset CER values**, so each dataset contributes equally regardless of line or character count. The four displayed Baseer values average to 28.675%; the supplied 28.67% display is retained, while the unrounded recomputation is recorded in `baseer_server_benchmark.json`. The supported conclusion is therefore: **under this server protocol, Phoenix has lower CER on three of four datasets and a much lower macro CER at a fraction of the model size; Baseer__Nakba remains stronger on Omar and exact-line accuracy.**
132
+
133
+ ## Same-protocol diagnostic comparison
134
+
135
+ All models below were decoded on the same pre-cropped line images and raw references with Kraken greedy decoding, without an LM or normalization. Each cell is **CER / word accuracy**.
136
+
137
+ | Development guard | Lines | Original Muharaf | exp6 | exp8 | **exp9** |
138
+ |---|---:|---:|---:|---:|---:|
139
+ | Agapet | 991 | 33.15 / 16.62 | 23.82 / 24.71 | 20.08 / 33.58 | **9.47 / 67.40** |
140
+ | Omar | 1,143 | 9.26 / 60.08 | 26.88 / 21.25 | 12.06 / 53.90 | **8.91 / 63.95** |
141
+ | RASAM | 1,789 | 39.01 / 12.06 | 9.04 / 66.78 | 7.95 / 69.90 | **7.67 / 70.46** |
142
+ | Muharaf | 920 | 13.28 / **64.15** | 36.47 / 15.74 | 12.72 / 58.30 | **12.31** / 59.32 |
143
+ | **Four-domain unweighted macro** | **4,843** | 23.68 / 38.23 | 24.05 / 32.12 | 13.20 / 53.92 | **9.59 / 65.28** |
144
 
145
+ Under this exact diagnostic protocol, exp9 had the lowest CER on **4/4** comparable guards and the highest word accuracy on **3/4**. Its macro CER was 9.59% versus 23.68% for the original Muharaf model, a 59.5% relative reduction in character errors. On the Muharaf guard itself, exp9 had lower CER (12.31% vs 13.28%) but lower word accuracy (59.32% vs 64.15%); both metrics are reported to avoid hiding the trade-off.
146
 
147
+ The 369-line TariMa manuscript guard is excluded from this macro because it was independent for exp9 but not for exp6/exp8. Its scores are retained in `BENCHMARKS.md` as a non-ranked diagnostic.
 
 
 
 
 
 
 
148
 
149
+ This table is a development diagnostic, not an independent final test: some of the guard sets participated in exp9 selection. Its purpose is to compare `muharaf_rec_best`, exp6, exp8, and exp9 under one raw-reference protocol. It must not be mixed with published Muharaf numbers obtained from a different split, codec, or normalization policy.
150
 
151
+ ## Training data
152
 
153
+ exp9 used document-aware training/replay from five sources:
154
+
155
+ | Dataset | Role | Recorded license |
156
  |---|---|---|
157
+ | Muharaf public line images | Archival handwriting | CC BY-NC-SA 2.0 |
158
+ | RASAM | Maghrebi manuscripts | Apache-2.0 in the local dataset repository |
159
+ | TariMa | Maghrebi/historical manuscripts | Apache-2.0 |
160
+ | Agapet SA-418 | 13th-century historical manuscript | CC BY 4.0 |
161
+ | Omar Al-Saleh manuscripts | Diverse archival documents | CC BY 4.0; access-gated at download time |
162
 
163
+ Training and guard splits were separated at document or manuscript level where the source allowed it. The Agapet sealed manuscripts and the eleven Omar sealed documents did not enter training.
164
 
165
+ Because Muharaf is CC BY-NC-SA 2.0, this release uses the same non-commercial ShareAlike license as the conservative publication choice. Users are responsible for checking whether their intended use and jurisdiction are compatible with every upstream dataset license.
166
+
167
+ ## Intended use
168
+
169
+ - Research and non-commercial transcription assistance for Arabic manuscripts and archival documents.
170
+ - Producing initial transcriptions for expert review.
171
+ - Generating multiple CTC candidates for a human-in-the-loop workflow.
172
+ - Use with PAGE-XML line polygons when layout segmentation is supplied externally.
173
+
174
+ ## Out-of-scope use
175
 
176
+ - Fully automatic scholarly editions without expert review.
177
+ - Claims of universal accuracy across all Arabic scripts, periods, or image conditions.
178
+ - Automatic attribution of text to a unique source without a separate retrieval/attribution layer.
179
+ - Commercial use without a separate legal review of the training-data obligations.
180
+ - Treating decoder scores as calibrated probabilities of correctness.
181
 
182
+ ## Inference
183
 
184
+ Install a Kraken version compatible with PyTorch 2.4, then use the model as a Kraken recognition model. A typical page command is:
185
 
186
+ ```bash
187
+ kraken -i page.jpg output.txt segment ocr -m model.mlmodel
 
 
 
 
 
188
  ```
189
 
190
+ For complex pages, the recommended Athar workflow supplies PAGE-XML polygons and runs recognition on the human-defined line regions. This avoids mixing segmentation failure with recognition error.
191
+
192
+ ## Language-model policy in Athar
193
+
194
+ The published CER/WER values are greedy recognition values. The complete Athar system can additionally:
195
+
196
+ - preserve the raw visual reading for every line;
197
+ - generate up to eight visual candidates;
198
+ - rerank candidates with a small local character n-gram model;
199
+ - apply the general/Maghrebi LM automatically only in a conservative confidence band;
200
+ - show an archival LM as advisory evidence instead of silently replacing the text;
201
+ - protect numbers and punctuation from destructive LM changes.
202
+
203
+ On a 1,789-line RASAM development guard, the deployed conservative general-LM policy produced only a small change (about −0.003 CER points and −0.141 WER points). The large exp9 gains therefore come from the visual recognizer, not from post-hoc language correction.
204
 
205
+ ## Athar system capabilities
 
206
 
207
+ The `.mlmodel` file is one component of the larger architecture. The application adds:
208
 
209
+ - visual and language-ranked alternatives;
210
+ - source retrieval with unique/ambiguous attribution states;
211
+ - a review queue with reasons;
212
+ - optional external-LLM suggestions labelled as advisory;
213
+ - auditable human accept/edit/reject decisions;
214
+ - PAGE-XML and TEI export;
215
+ - a training package that excludes unreviewed automatic output by default.
216
 
217
+ These are system capabilities, not properties encoded inside the model weights.
218
 
219
+ ## Limitations
220
+
221
+ - Raw CER remains 17.86% on the difficult Agapet sealed set; human review is still necessary.
222
+ - TariMa sealed performance is 0.33 CER points worse than exp8.
223
+ - The model is optimized for Arabic handwriting, not modern printed OCR.
224
+ - Layout/segmentation quality can dominate full-page performance.
225
+ - The 81-symbol codec does not cover every possible Arabic-script or Latin character.
226
+ - Confidence values and beam scores are not globally calibrated probabilities.
227
+ - The same-protocol Muharaf comparison in this repository is a development comparison, not a new sealed benchmark.
228
+
229
+ ## Reproducibility
230
+
231
+ - Model SHA-256: `2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea`
232
+ - Python used in the verified local environment: 3.10.11
233
+ - PyTorch: 2.4.1+cu121
234
+ - Evaluation decoder: greedy, without LM
235
+ - Raw and Arabic-normalized metrics are stored separately; the model card reports raw metrics unless explicitly labelled otherwise.
236
+
237
+ See `BENCHMARKS.md`, `benchmark_summary.json`, `baseer_server_benchmark.json`, `DATA_AND_LICENSES.md`, and `metadata.json` in this repository for details.
238
+
239
+ ## Citation
240
+
241
+ Until a final paper record is available, cite the project and this model version as:
242
+
243
+ ```bibtex
244
+ @misc{athar_htr_exp9_2026,
245
+ title = {Phoenix: Arabic Manuscript HTR Model},
246
+ author = {Athar Project Team},
247
+ year = {2026},
248
+ note = {Internal checkpoint exp9; Kraken CNN-BiLSTM-CTC model; SHA-256 2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea}
249
+ }
250
+ ```
251
 
252
+ ## Ethical and scholarly note
253
 
254
+ The output is a research aid, not an authoritative edition. Manuscript transcription and source attribution require domain expertise. Preserve the image, the raw reading, the evidence trail, and the researcher's final decision.
 
 
 
SHA256SUMS CHANGED
@@ -1,4 +1,4 @@
1
  2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea model.mlmodel
2
  7184ae01c9f4a248d80ddeb84d852a8c21b8d7b213ca7619433edf4b18bf2fe7 benchmark_summary.json
3
- 2592cdb457098fbac54bc80c37799e80af6ce09c5491d5a65be32d1bc3bae3c9 metadata.json
4
  1cc91f05e0ff7e5594b13029a0b82d0fa2b78de3dff1a16160d99ed61441f039 baseer_server_benchmark.json
 
1
  2896fef9d9665cbb82fba8faa3bf0c628ac6a5cc62eb678f40c707db833aebea model.mlmodel
2
  7184ae01c9f4a248d80ddeb84d852a8c21b8d7b213ca7619433edf4b18bf2fe7 benchmark_summary.json
3
+ d943903a68dd40859b42a66f244d8441b022a8b051a8b1d0716781b7ef9919e4 metadata.json
4
  1cc91f05e0ff7e5594b13029a0b82d0fa2b78de3dff1a16160d99ed61441f039 baseer_server_benchmark.json
metadata.json CHANGED
@@ -12,7 +12,7 @@
12
  "python": "3.10.11",
13
  "pytorch": "2.4.1+cu121",
14
  "created_date": "2026-08-11",
15
- "release_status": "uploaded_private_pending_publication_review",
16
  "license": "CC BY-NC-SA 2.0 (conservative choice due to upstream Muharaf training data)",
17
  "sealed_release_evidence": {
18
  "lines": 22442,
 
12
  "python": "3.10.11",
13
  "pytorch": "2.4.1+cu121",
14
  "created_date": "2026-08-11",
15
+ "release_status": "published_public_2026-08-17",
16
  "license": "CC BY-NC-SA 2.0 (conservative choice due to upstream Muharaf training data)",
17
  "sealed_release_evidence": {
18
  "lines": 22442,