Tet3u Jackrong commited on
Commit
a12229e
·
0 Parent(s):

Duplicate from Jackrong/Qwen3.5-9B-Gemini-3.1-Pro-Reasoning-Distill-GGUF

Browse files
.gitattributes ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ Qwen3.5-9B.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
37
+ Qwen3.5-9B.Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
38
+ Qwen3.5-9B.Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
39
+ Qwen3.5-9B.Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
40
+ Qwen3.5-9B.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
41
+ Qwen3.5-9B.Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
42
+ Qwen3.5-9B.Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
43
+ Qwen3.5-9B.Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
44
+ Qwen3.5-9B.Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
45
+ Qwen3.5-9B.Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
46
+ Qwen3.5-9B.BF16-mmproj.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3.5-9B.BF16-mmproj.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:485829f9e5cb0cb13ca0a55096baa4138a086534bdf62e31f932b45e9d0b1edc
3
+ size 921704576
Qwen3.5-9B.Q3_K_M.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f992f07be3128b46eb40de73ef490b111c6c9c2c3b8b06060c9d2c24298176bd
3
+ size 4623520640
Qwen3.5-9B.Q4_K_M.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:71b7a83558ea6032ac53a30675441b49a55d54127f552dccdfe4343c868ced77
3
+ size 5737239424
Qwen3.5-9B.Q5_K_M.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f0839dbaae005c6c92094abd89123e179a1de68f5367e973ccc67841d61bdc53
3
+ size 6523671424
Qwen3.5-9B.Q6_K.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:892b06a7ec2d3964fec873f5ea1bae76c505ea303322845f544b670b49856fc3
3
+ size 7359255424
Qwen3.5-9B.Q8_0.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af590234581539020499db105b8c219a56a3a3ff88d26c47bde1b7f9b6ccca5a
3
+ size 9527497600
README.md ADDED
@@ -0,0 +1,115 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language:
3
+ - en
4
+ - zh
5
+ - ko
6
+ license: apache-2.0
7
+ base_model: Qwen/Qwen3.5-9B
8
+ tags:
9
+ - unsloth
10
+ - qwen
11
+ - qwen3.5
12
+ - reasoning
13
+ - chain-of-thought
14
+ - distillation
15
+ - Dense
16
+ pipeline_tag: text-generation
17
+ datasets:
18
+ - Jackrong/Qwen3.5-reasoning-700x
19
+ - Roman1111111/gemini-3.1-pro-hard-high-reasoning
20
+ ---
21
+
22
+
23
+ # 🌟 Qwen3.5-9B-Gemini-3.1-Pro-Reasoning-Distill
24
+
25
+ ## 💡 Model Introduction
26
+ **Qwen3.5-9B-Gemini-3.1-Pro-Reasoning-Distill** is a reasoning model fine-tuned on top of **Qwen3.5-9B**.
27
+ The model is primarily optimized through high-density reasoning distillation sourced from **Gemini 3.1**, while also incorporating additional reasoning traces distilled from **Qwen3.5-27B** and a broader **Gemini 3.0 Pro** reasoning corpus.
28
+
29
+ Through Supervised Fine-Tuning focused on structured analytical behavior, this model aims to reshape the base model’s reasoning style into a more coherent, better-organized, and higher-density Chain-of-Thought (CoT) pattern.
30
+ It is especially designed to improve decomposition, planning, abstraction, and response cleanliness on complex multi-step tasks.
31
+
32
+ ---
33
+
34
+ ## 🧠 Example of Learned Reasoning Scaffold
35
+
36
+ This model inherits a more structured reasoning style influenced by **Gemini 3.1-style analytical planning**.
37
+ Compared with more loosely exploratory reasoning patterns, this model tends to organize the problem before answering:
38
+
39
+ ```text
40
+ My Thought Process / My Analysis of the problem:
41
+
42
+ 1. Restate the task and identify the true objective.
43
+ 2. Abstract the problem into a higher-level reasoning frame.
44
+ 3. Identify the key mechanism, failure mode, or constraint.
45
+ 4. Separate likely misconceptions from the actual core issue.
46
+ 5. Plan the structure of the final response.
47
+ 6. Deliver a cleaner, more direct, and higher-density answer.
48
+ .
49
+ .
50
+ .
51
+ ```
52
+ ---
53
+
54
+ ## 🗺️ Training Pipeline Overview
55
+ ```text
56
+ Base Model (Qwen3.5-9B)
57
+
58
+
59
+ Supervised Fine-Tuning (SFT) + LoRA + Reasoning Distillation
60
+ (Response-Only Training masked on "<|im_start|>assistant\n<think>")
61
+
62
+
63
+ Final Model Text Only (Jackrong/Qwen3.5-9B-Gemini-3.1-Pro-Reasoning-Distill)
64
+ ```
65
+
66
+ ## 📋 Stage Details
67
+
68
+ ### 🔹 Supervised Fine-Tuning (SFT)
69
+ - **Objective:** Objective: To inject reasoning behavior into Qwen3.5-9B and strengthen its performance on complex analytical tasks requiring decomposition and multi-step inference.
70
+ - **Method:** The model is trained on distilled reasoning traces collected from stronger teacher-style reasoning sources, with the goal of transferring cleaner analytical structure, stronger planning habits, and more stable task-solving behavior.
71
+ - **Target Behavior:** Compared with a standard instruct model, the tuned model is expected to respond with more deliberate reasoning organization, reduced shallow guessing, and stronger cross-domain analytical consistency.
72
+
73
+ ### 📚 All Datasets Used
74
+ The dataset consists of multiple reasoning distillation sources:
75
+
76
+ | Dataset Name | Description / Purpose |
77
+ |--------------|-----------------------|
78
+ | [Roman1111111/gemini-3.1-pro-hard-high-reasoning](https://huggingface.co/datasets/Roman1111111/gemini-3.1-pro-hard-high-reasoning) | Primary high-quality reasoning source used to shape structured analytical style, planning behavior, and dense CoT patterns. |
79
+ | [Jackrong/Qwen3.5-reasoning-700x](https://huggingface.co/datasets/Jackrong/Qwen3.5-reasoning-700x) | Provides additional Qwen-family reasoning trajectories distilled from Qwen3.5-27B, improving style stability and complementary reasoning diversity. |
80
+ | [Roman1111111/gemini-3-pro-10000x-hard-high-reasoning](https://huggingface.co/datasets/Roman1111111/gemini-3-pro-10000x-hard-high-reasoning) | A broader multi-domain reasoning corpus used to enhance coverage across mathematics, systems, science, law, medicine, finance, and adversarial reasoning tasks. |
81
+
82
+ ### 📊 Approximate Domain Composition (Approx|Samples|Share)
83
+
84
+ | Domain | Samples | Share |
85
+ |--------------|--------:|------:|
86
+ | Mathematics / Logic | 3947 | 28.5% |
87
+ | Computer Science / Programming / Systems | 3019 | 21.8% |
88
+ | Security / Adversarial Reasoning | 1551 | 11.2% |
89
+ | Physics / Astronomy / Engineering | 1482 | 10.7% |
90
+ | Law / Philosophy / Humanities | 1191 | 8.6% |
91
+ | Biology / Medicine | 817 | 5.9% |
92
+ | Finance / Economics | 679 | 4.9% |
93
+ | Chemistry / Materials | 540 | 3.9% |
94
+ | Applied / Social Systems (Urban Planning, Traffic, Supply Chain, etc.) | 360 | 2.6% |
95
+ | Other | 264 | 1.9% |
96
+
97
+ ⚠️ **Distillation & Task-Specific Fine-Tuning Effects:** This model has been distilled and further fine-tuned on top of the base model for reasoning-oriented tasks. These techniques may improve performance on certain specialized tasks, but they can also influence the model’s generalization ability in broader scenarios and may lead to partial forgetting of some pretraining knowledge. The extent of these effects depends on factors such as the quality, scale, and distribution of the datasets used during distillation and fine-tuning. As a result, the model’s behavior may differ from the base model across different tasks or application contexts. Users are encouraged to evaluate the model according to their specific requirements before deployment. Thank you for your understanding~
98
+
99
+
100
+
101
+ ## 🌟 Core Skills & Capabilities
102
+ 1. **Structured Analytical Reasoning:** The model is optimized to first identify the real task structure before generating an answer, rather than relying on shallow immediate completion.
103
+ 2. **Improved Multi-Step Planning:** It performs more reliably on tasks requiring decomposition, constraint tracking, sequential planning, and trade-off analysis.
104
+ 3. **Cross-Domain Reasoning Strength:** The training corpus provides broad reasoning coverage across math, programming, systems, physics, law, medicine, finance, chemistry, and applied domains.
105
+ 4. **Security & Adversarial Awareness:** A dedicated portion of the distilled data includes adversarial, attack-defense, and failure-mode reasoning tasks, improving robustness in difficult prompts.
106
+ 5. **Compact but Strong Footprint:** Built on a 9B base, the model aims to deliver significantly denser reasoning behavior and cleaner analytical output than a generic base instruct model of similar size.
107
+
108
+ ## ⚠️ Limitations & Intended Use
109
+ - **Hallucination Risk:** Although reasoning behavior is improved, the model remains an autoregressive LLM and may still hallucinate niche facts, citations, or unverifiable real-world details.
110
+ - **Reasoning Style Bias:** Because the model is tuned for analytical depth, it may sometimes produce longer or more structured answers than necessary for very simple prompts.
111
+ - **Teacher-Style Distillation Bias:** Some response behaviors reflect the reasoning style of the teacher traces used during distillation, rather than purely native behavior emerging from the base model itself.
112
+ - **Preview Version Notice:** As a relatively specialized distilled reasoning model, surrounding inference templates, prompt formatting strategies, and ecosystem integrations may still require tuning. Users may encounter occasional compatibility differences depending on runtime or deployment stack.
113
+
114
+ ## 🙏 Acknowledgements
115
+ Special thanks to the **Qwen** team for the strong base architecture, and to the broader open-source ecosystem for enabling efficient reasoning distillation workflows. We also acknowledge the value of the distilled reasoning corpora derived from **Gemini 3.1 Pro**, **Qwen3.5**, and **Gemini 3 Pro**, which made this model possible.
config.json ADDED
@@ -0,0 +1,113 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5ForConditionalGeneration"
4
+ ],
5
+ "torch_dtype": "bfloat16",
6
+ "eos_token_id": 248046,
7
+ "image_token_id": 248056,
8
+ "model_name": "qwen/Qwen3.5-9B",
9
+ "model_type": "qwen3_5",
10
+ "pad_token_id": 248044,
11
+ "text_config": {
12
+ "attention_bias": false,
13
+ "attention_dropout": 0.0,
14
+ "attn_output_gate": true,
15
+ "bos_token_id": null,
16
+ "torch_dtype": "bfloat16",
17
+ "eos_token_id": 248044,
18
+ "full_attention_interval": 4,
19
+ "head_dim": 256,
20
+ "hidden_act": "silu",
21
+ "hidden_size": 4096,
22
+ "initializer_range": 0.02,
23
+ "intermediate_size": 12288,
24
+ "layer_types": [
25
+ "linear_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "full_attention",
29
+ "linear_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "full_attention",
33
+ "linear_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "full_attention",
37
+ "linear_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "full_attention",
41
+ "linear_attention",
42
+ "linear_attention",
43
+ "linear_attention",
44
+ "full_attention",
45
+ "linear_attention",
46
+ "linear_attention",
47
+ "linear_attention",
48
+ "full_attention",
49
+ "linear_attention",
50
+ "linear_attention",
51
+ "linear_attention",
52
+ "full_attention",
53
+ "linear_attention",
54
+ "linear_attention",
55
+ "linear_attention",
56
+ "full_attention"
57
+ ],
58
+ "linear_conv_kernel_dim": 4,
59
+ "linear_key_head_dim": 128,
60
+ "linear_num_key_heads": 16,
61
+ "linear_num_value_heads": 32,
62
+ "linear_value_head_dim": 128,
63
+ "mamba_ssm_dtype": "float32",
64
+ "max_position_embeddings": 262144,
65
+ "mlp_only_layers": [],
66
+ "model_type": "qwen3_5_text",
67
+ "mtp_num_hidden_layers": 1,
68
+ "mtp_use_dedicated_embeddings": false,
69
+ "num_attention_heads": 16,
70
+ "num_hidden_layers": 32,
71
+ "num_key_value_heads": 4,
72
+ "pad_token_id": null,
73
+ "partial_rotary_factor": 0.25,
74
+ "rms_norm_eps": 1e-06,
75
+ "rope_parameters": {
76
+ "mrope_interleaved": true,
77
+ "mrope_section": [
78
+ 11,
79
+ 11,
80
+ 10
81
+ ],
82
+ "partial_rotary_factor": 0.25,
83
+ "rope_theta": 10000000,
84
+ "rope_type": "default"
85
+ },
86
+ "tie_word_embeddings": false,
87
+ "use_cache": true,
88
+ "vocab_size": 248320
89
+ },
90
+ "tie_word_embeddings": false,
91
+ "unsloth_version": "2026.3.4",
92
+ "use_cache": false,
93
+ "video_token_id": 248057,
94
+ "vision_config": {
95
+ "deepstack_visual_indexes": [],
96
+ "depth": 27,
97
+ "torch_dtype": "bfloat16",
98
+ "hidden_act": "gelu_pytorch_tanh",
99
+ "hidden_size": 1152,
100
+ "in_channels": 3,
101
+ "initializer_range": 0.02,
102
+ "intermediate_size": 4304,
103
+ "model_type": "qwen3_5",
104
+ "num_heads": 16,
105
+ "num_position_embeddings": 2304,
106
+ "out_hidden_size": 4096,
107
+ "patch_size": 16,
108
+ "spatial_merge_size": 2,
109
+ "temporal_patch_size": 2
110
+ },
111
+ "vision_end_token_id": 248054,
112
+ "vision_start_token_id": 248053
113
+ }