{ "manifest_schema": "0.1.2", "repo": "litert-community/SmolLM3-3B", "generated": "2026-09-21", "generator": "make_manifest.py", "model": { "display_name": "SmolLM3-3B", "base_model": "HuggingFaceTB/SmolLM3-3B", "architecture": "Fully-open 3B decoder with GQA and a NoPE attention schedule (SmolLM3ForCausalLM, rotary disabled every 4th layer), multilingual, long-context trained", "parameters_b": 3, "license": "apache-2.0", "context_length": 4096, "capabilities": { "vision": false, "audio": false, "thinking": { "declared": true, "channel": { "start": "\n", "end": "\n" } }, "channels": [ { "name": "thought", "start": "\n", "end": "\n" } ] } }, "variants": [ { "file": "SmolLM3-3B.litertlm", "sha256": "f9aa844c27f5ccfae9cb701c7ca63f2faba501ed187aacc85044b6200df15a37", "size_bytes": 3109328112, "sections": [ { "type": "LlmMetadataProto", "size_bytes": 3318 }, { "type": "HF_Tokenizer_Zlib", "size_bytes": 2608153 }, { "type": "TFLiteModel", "size_bytes": 3106673904, "model_type": "tf_lite_prefill_decode" } ], "quantization": "int4 (recipe not stated on card)", "backends": [ "cpu", "gpu" ], "default_backend": "cpu", "measured": [ { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "gpu", "runtime": "litert_lm_advanced_main v0.16.0, LiteRT CL delegate", "prompt_tokens": 211, "decode_tokens": 295, "prefill_tps": 314.0, "decode_tps": 14.38, "ttft_s": 0.74, "load_s": 4.4, "peak_memory_mb": 704, "runs": 1, "date": "2026-08-24", "source": "S4 GPU gate, 205-token benchmark, single run; full delegation (2930 ops across 2 subgraphs); generation gate answered the probe correctly. No same-device CPU control was taken, so this is a GPU speed, not a measured win over CPU." }, { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "cpu", "runtime": "litert_lm_advanced_main v0.16.0 (XNNPACK CPU)", "prompt_tokens": 442, "decode_tokens": 912, "prefill_tps": "124.5-147.9", "decode_tps": "9.22-9.81", "ttft_s": "3.09-3.66", "load_s": 9.1, "peak_memory_mb": 4118, "cache": "no", "runs": 2, "date": "2026-09-05", "source": "S7 Android backfill: same-device CPU control - identical binary, prompt file and flags, only --backend differs; 205-token prompt with --benchmark, 2 cold runs (compile/weight caches deleted between runs, device cooled below 42 C before each run); same-device control for an existing GPU row on this handset" } ], "known_issues": [ "File is present in the repo tree (3.11 GB) but not documented anywhere on the model card" ] }, { "file": "SmolLM3-3B_q4_block32_ekv4096.litertlm", "sha256": "119fc1f266c445e9ffbc8c16695eb325b629da3fd464e73a75e5db14ef3926c3", "size_bytes": 2002257840, "sections": [ { "type": "LlmMetadataProto", "size_bytes": 2963 }, { "type": "HF_Tokenizer_Zlib", "size_bytes": 2608153 }, { "type": "TFLiteModel", "size_bytes": 1733852544, "model_type": "tf_lite_prefill_decode" }, { "type": "TFLiteModel", "size_bytes": 265750448, "model_type": "tf_lite_embedder" } ], "quantization": "int4 weights - blockwise (block 32) + OCTAV optimal-clipping, symmetric; embedding INT8", "backends": [ "cpu", "gpu" ], "default_backend": "gpu", "recommended": [ { "platform": "android", "device_class": "flagship", "backend": "gpu", "reason": "the Android path verified on a Galaxy S26 (S4 gate 2026-08-24): full delegation (2784 ops across 2 subgraphs) and a correct answer to the probe; 17.9 tok/s decode, engine init 9.6 s, peak 1111 MB; the repo's other S26-verified file (SmolLM3-3B.litertlm) decodes 14.4 tok/s on the same handset. No same-device CPU control was taken, so this is the verified choice for the class, not a measured win over CPU" }, { "platform": "android", "device_class": "midrange-2023+", "backend": "gpu", "reason": "verified on a Pixel 8a (Tensor G3, 8 GB): full OpenCL delegation (1476/1476/1308/1308 nodes) and a correct answer in the 2026-08-17 re-gate; no speed row and no same-device CPU control, so this is the verified GPU path for the class, not a measured win" } ], "requirements": { "platform_notes": [ "Gallery import needs package com.google.ai.edge.gallery 1.0.15+ (older 1.0.x builds reject .litertlm); Gallery v1.0.16+ can import litert-lm models directly from Hugging Face inside the app", "Embedding externalized into its own bundle section so the main weights section stays under the iOS ~2 GiB single-mmap limit", "On a Pixel 8a (Tensor G3, 8 GB) litert_lm_main runs the graph entirely on the OpenCL delegate (1476/1476 prefill, 1308/1308 decode nodes, zero rejected ops) and answers correctly" ] }, "measured": [ { "device": "Apple M4 Max", "os": "macOS", "backend": "cpu", "runtime": "litert-lm benchmark (litert-lm 0.15.0)", "prompt_tokens": 256, "decode_tokens": 256, "prefill_tps": 141, "decode_tps": 24.1, "ttft_s": 2.14, "max_num_tokens": 4096, "runs": 3, "date": "2026-08-24", "source": "model card Performance table (cardbench harness)" }, { "device": "Apple M4 Max", "os": "macOS", "backend": "gpu", "runtime": "litert-lm benchmark (litert-lm 0.15.0)", "prompt_tokens": 256, "decode_tokens": 256, "prefill_tps": 1354, "decode_tps": 93.2, "ttft_s": 0.21, "max_num_tokens": 4096, "runs": 3, "date": "2026-08-24", "source": "model card Performance table (cardbench harness)" }, { "device": "iPhone 17 Pro", "os": "iOS 27.0", "backend": "gpu", "runtime": "LiteRTDemo harness (Metal GPU backend)", "prefill_tps": 30.8, "decode_tps": 22.5, "ttft_s": 0.63, "runs": 1, "date": "2026-08-24", "source": "model card Performance table (cardbench harness)" }, { "device": "Galaxy S26 (SM-S942Q, Qualcomm SM8850, Adreno)", "backend": "gpu", "runtime": "litert_lm_advanced_main v0.16.0, LiteRT CL delegate", "prompt_tokens": 202, "decode_tokens": 358, "prefill_tps": 268.3, "decode_tps": 17.9, "ttft_s": 0.81, "load_s": 9.6, "peak_memory_mb": 1111, "runs": 1, "date": "2026-08-24", "source": "S4 GPU gate, 205-token benchmark, single run; full delegation (2784 ops across 2 subgraphs); generation gate answered the probe correctly. No same-device CPU control was taken, so this is a GPU speed, not a measured win over CPU." }, { "device": "Pixel 8a (Tensor G3, Mali-G715, 8 GB)", "backend": "gpu", "runtime": "litert_lm_main v0.16.0 (tag-pinned) with the v0.16.0 prebuilt android_arm64 libraries, OpenCL", "runs": 1, "date": "2026-08-17", "source": "Android GPU re-gate: full delegation (1476/1476 prefill / 1308/1308 decode nodes, zero rejections); correct answer (Burj Khalifa, 828 m). Gate only - no speed row." } ] } ] }