harmya-modal commited on
Commit
dae6d31
·
0 Parent(s):

Add GLM-5.3-Flash DFlash draft model

Browse files
Files changed (5) hide show
  1. .gitattributes +35 -0
  2. LICENSE +21 -0
  3. README.md +64 -0
  4. config.json +55 -0
  5. model.safetensors +3 -0
.gitattributes ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Z.AI Co., Ltd
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
README.md ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ pipeline_tag: text-generation
3
+ library_name: transformers
4
+ base_model:
5
+ - zai-org/GLM-5.3-Flash
6
+ license: mit
7
+ license_link: LICENSE
8
+ inference: false
9
+ tags:
10
+ - dflash
11
+ - speculative-decoding
12
+ - speculative-decoding-draft
13
+ - block-diffusion
14
+ - draft-model
15
+ - glm
16
+ - glm-5.3
17
+ - sglang
18
+ ---
19
+
20
+ # GLM-5.3-Flash-DFlash
21
+
22
+ [Paper](https://arxiv.org/abs/2602.06036) | [Github](https://github.com/z-lab/dflash) | [Blog](https://z-lab.ai/projects/dflash)
23
+
24
+ This repository contains a DFlash draft model for [`zai-org/GLM-5.3-Flash`](https://huggingface.co/zai-org/GLM-5.3-Flash). It is not a standalone language model. It is intended to be paired with the target model in a speculative decoding server.
25
+
26
+ DFlash uses a lightweight block diffusion draft model to propose multiple tokens in parallel. The target model verifies those proposals, improving serving throughput while preserving the target model's output distribution.
27
+
28
+ ## Quick Start
29
+
30
+ GLM-5.3-Flash needs the DFlash capture hooks in the `glm5_next` model, available on SGLang main. An example deployment is:
31
+
32
+ ```bash
33
+ python -m sglang.launch_server \
34
+ --model-path zai-org/GLM-5.3-Flash \
35
+ --tp-size 4 \
36
+ --trust-remote-code \
37
+ --speculative-algorithm DFLASH \
38
+ --speculative-draft-model-path modal-labs/GLM-5.3-Flash-DFlash \
39
+ --speculative-dflash-block-size 8 \
40
+ --speculative-draft-model-quantization unquant \
41
+ --speculative-draft-attention-backend trtllm_mha \
42
+ --speculative-draft-kv-cache-dtype fp8_e4m3 \
43
+ --host 0.0.0.0 \
44
+ --port 30000
45
+ ```
46
+
47
+ Keep the draft model unquantized. Quantizing it lowers the accept length.
48
+
49
+ ## License
50
+
51
+ Distributed under the [MIT License](LICENSE), inherited from the target model.
52
+
53
+ ## Citation
54
+
55
+ If you find DFlash useful, please cite the original paper:
56
+
57
+ ```bibtex
58
+ @article{chen2026dflash,
59
+ title = {{DFlash: Block Diffusion for Flash Speculative Decoding}},
60
+ author = {Chen, Jian and Liang, Yesheng and Liu, Zhijian},
61
+ journal = {arXiv preprint arXiv:2602.06036},
62
+ year = {2026}
63
+ }
64
+ ```
config.json ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "DFlash2DraftModel"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "dflash_config": {
8
+ "block_size": 8,
9
+ "conv_group_size": 16,
10
+ "conv_kernel_size": 2,
11
+ "mask_token_id": 154856,
12
+ "output_multiplier": 1.0,
13
+ "selector_rank": 256,
14
+ "selector_top_k": 16,
15
+ "target_layer_ids": [
16
+ 23,
17
+ 27,
18
+ 31,
19
+ 35,
20
+ 39,
21
+ 43
22
+ ]
23
+ },
24
+ "dtype": "bfloat16",
25
+ "head_dim": 128,
26
+ "hidden_act": "silu",
27
+ "hidden_size": 4096,
28
+ "initializer_range": 0.02,
29
+ "intermediate_size": 12288,
30
+ "is_causal": true,
31
+ "layer_types": [
32
+ "sliding_attention",
33
+ "sliding_attention",
34
+ "sliding_attention",
35
+ "sliding_attention",
36
+ "sliding_attention",
37
+ "sliding_attention"
38
+ ],
39
+ "max_position_embeddings": 1048576,
40
+ "model_type": "qwen3",
41
+ "num_attention_heads": 32,
42
+ "num_hidden_layers": 6,
43
+ "num_key_value_heads": 8,
44
+ "num_target_layers": 45,
45
+ "rms_norm_eps": 1e-05,
46
+ "rope_parameters": {
47
+ "rope_theta": 2000000.0,
48
+ "rope_type": "default"
49
+ },
50
+ "sliding_window": 4096,
51
+ "tie_word_embeddings": false,
52
+ "transformers_version": "5.7.0",
53
+ "use_sliding_window": true,
54
+ "vocab_size": 154880
55
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3860c574465b7c7523896a22a3cc83dc4b9c14e019a6755061de2b8a4b061539
3
+ size 2778461544