mlboydaisuke commited on
Commit
ff4fc37
·
verified ·
1 Parent(s): 7c5ec04

Add litertlm_manifest.json — machine-readable deployment manifest (variant selection, backend recommendations, measured performance)

Browse files
Files changed (1) hide show
  1. litertlm_manifest.json +134 -0
litertlm_manifest.json ADDED
@@ -0,0 +1,134 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "manifest_schema": "0.1.0",
3
+ "repo": "litert-community/SmolLM3-3B",
4
+ "generated": "2026-08-24",
5
+ "generator": "make_manifest.py",
6
+ "model": {
7
+ "display_name": "SmolLM3-3B",
8
+ "base_model": "HuggingFaceTB/SmolLM3-3B",
9
+ "architecture": "Fully-open 3B decoder with GQA and a NoPE attention schedule (SmolLM3ForCausalLM, rotary disabled every 4th layer), multilingual, long-context trained",
10
+ "parameters_b": 3,
11
+ "license": "apache-2.0",
12
+ "context_length": 4096,
13
+ "capabilities": {
14
+ "vision": false,
15
+ "audio": false,
16
+ "thinking": {
17
+ "declared": false
18
+ }
19
+ }
20
+ },
21
+ "variants": [
22
+ {
23
+ "file": "SmolLM3-3B.litertlm",
24
+ "sha256": "a34da46f5697896c98a4d3b3dacddbc50487f8e35432b37b72e4fe38ed264bb4",
25
+ "size_bytes": 3108978688,
26
+ "sections": [
27
+ {
28
+ "type": "LlmMetadataProto",
29
+ "size_bytes": 268
30
+ },
31
+ {
32
+ "type": "SP_Tokenizer",
33
+ "size_bytes": 2260840
34
+ },
35
+ {
36
+ "type": "TFLiteModel",
37
+ "size_bytes": 3106673904,
38
+ "model_type": "tf_lite_prefill_decode"
39
+ }
40
+ ],
41
+ "quantization": "int4 (recipe not stated on card)",
42
+ "backends": [
43
+ "cpu"
44
+ ],
45
+ "default_backend": "cpu",
46
+ "measured": [],
47
+ "known_issues": [
48
+ "File is present in the repo tree (3.11 GB) but not documented anywhere on the model card"
49
+ ]
50
+ },
51
+ {
52
+ "file": "SmolLM3-3B_q4_block32_ekv4096.litertlm",
53
+ "sha256": "38d7bf55e243f5a95d8b16c21b1f7ef7debe6ce6bb2ddb66eb54f4bc2be8e6eb",
54
+ "size_bytes": 2002257840,
55
+ "sections": [
56
+ {
57
+ "type": "LlmMetadataProto",
58
+ "size_bytes": 142
59
+ },
60
+ {
61
+ "type": "HF_Tokenizer_Zlib",
62
+ "size_bytes": 2608153
63
+ },
64
+ {
65
+ "type": "TFLiteModel",
66
+ "size_bytes": 1733852544,
67
+ "model_type": "tf_lite_prefill_decode"
68
+ },
69
+ {
70
+ "type": "TFLiteModel",
71
+ "size_bytes": 265750448,
72
+ "model_type": "tf_lite_embedder"
73
+ }
74
+ ],
75
+ "quantization": "int4 weights - blockwise (block 32) + OCTAV optimal-clipping, symmetric; embedding INT8",
76
+ "backends": [
77
+ "cpu",
78
+ "gpu"
79
+ ],
80
+ "default_backend": "gpu",
81
+ "requirements": {
82
+ "platform_notes": [
83
+ "Gallery import needs package com.google.ai.edge.gallery 1.0.15+ (older 1.0.x builds reject .litertlm); Gallery v1.0.16+ can import litert-lm models directly from Hugging Face inside the app",
84
+ "Embedding externalized into its own bundle section so the main weights section stays under the iOS ~2 GiB single-mmap limit",
85
+ "On a Pixel 8a (Tensor G3, 8 GB) litert_lm_main runs the graph entirely on the OpenCL delegate (1476/1476 prefill, 1308/1308 decode nodes, zero rejected ops) and answers correctly"
86
+ ]
87
+ },
88
+ "measured": [
89
+ {
90
+ "device": "Apple M4 Max",
91
+ "os": "macOS",
92
+ "backend": "cpu",
93
+ "runtime": "litert-lm benchmark (litert-lm 0.15.0)",
94
+ "prompt_tokens": 256,
95
+ "decode_tokens": 256,
96
+ "prefill_tps": 141,
97
+ "decode_tps": 24.1,
98
+ "ttft_s": 2.14,
99
+ "max_num_tokens": 4096,
100
+ "runs": 3,
101
+ "date": "2026-08-24",
102
+ "source": "model card Performance table (cardbench harness)"
103
+ },
104
+ {
105
+ "device": "Apple M4 Max",
106
+ "os": "macOS",
107
+ "backend": "gpu",
108
+ "runtime": "litert-lm benchmark (litert-lm 0.15.0)",
109
+ "prompt_tokens": 256,
110
+ "decode_tokens": 256,
111
+ "prefill_tps": 1354,
112
+ "decode_tps": 93.2,
113
+ "ttft_s": 0.21,
114
+ "max_num_tokens": 4096,
115
+ "runs": 3,
116
+ "date": "2026-08-24",
117
+ "source": "model card Performance table (cardbench harness)"
118
+ },
119
+ {
120
+ "device": "iPhone 17 Pro",
121
+ "os": "iOS 27.0",
122
+ "backend": "gpu",
123
+ "runtime": "LiteRTDemo harness (Metal GPU backend)",
124
+ "prefill_tps": 30.8,
125
+ "decode_tps": 22.5,
126
+ "ttft_s": 0.63,
127
+ "runs": 1,
128
+ "date": "2026-08-24",
129
+ "source": "model card Performance table (cardbench harness)"
130
+ }
131
+ ]
132
+ }
133
+ ]
134
+ }