mario-rc commited on
Commit
98ee7f9
·
verified ·
1 Parent(s): 55a345e

Publish verified emotional RLAIF adapter and unified 20-model card

Browse files

Exact selected local artifact; family-specific license, training provenance and inference example. Numerical evaluation is intentionally excluded pending owner approval.

.gitattributes CHANGED
@@ -1,36 +1,3 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
  *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
36
  tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  *.safetensors filter=lfs diff=lfs merge=lfs -text
2
+ *.model filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
3
  tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Source: https://ai.google.dev/gemma/terms
2
+ Retrieved: 2026-09-16
3
+
4
+ The terms below apply to Gemma models listed in the Appendix at bottom of this page. For Gemma 4 terms, see the
5
+ Gemma 4 license
6
+ .
7
+ Last modified: April 1, 2026
8
+ By using, reproducing, modifying, distributing, performing or displaying any
9
+ portion or element of Gemma, Model Derivatives including via any Hosted Service,
10
+ (each as defined below) (collectively, the "
11
+ Gemma Services
12
+ ") or otherwise
13
+ accepting the terms of this Agreement, you agree to be bound by this Agreement.
14
+ Section 1: DEFINITIONS
15
+ 1.1 Definitions
16
+ (a) "
17
+ Agreement
18
+ " or "
19
+ Gemma Terms of Use
20
+ " means these terms and conditions
21
+ that govern the use, reproduction, Distribution or modification of the Gemma
22
+ Services and any terms and conditions incorporated by reference.
23
+ (b) "
24
+ Distribution
25
+ " or "
26
+ Distribute
27
+ " means any transmission, publication,
28
+ or other sharing of Gemma or Model Derivatives to a third party, including by
29
+ providing or making Gemma or its functionality available as a hosted service via
30
+ API, web access, or any other electronic or remote means ("
31
+ Hosted Service
32
+ ").
33
+ (c) "
34
+ Gemma
35
+ " means the set of machine learning language models, trained model
36
+ weights and parameters identified in the
37
+ Appendix
38
+ ,
39
+ regardless of the source that you obtained it from.
40
+ (d) "
41
+ Google
42
+ " means Google LLC.
43
+ (e) "
44
+ Model Derivatives
45
+ " means all (i) modifications to Gemma, (ii) works based
46
+ on Gemma, or (iii) any other machine learning model which is created by transfer
47
+ of patterns of the weights, parameters, operations, or Output of Gemma, to that
48
+ model in order to cause that model to perform similarly to Gemma, including
49
+ distillation methods that use intermediate data representations or methods based
50
+ on the generation of synthetic data Outputs by Gemma for training that model.
51
+ For clarity, Outputs are not deemed Model Derivatives.
52
+ (f) "
53
+ Output
54
+ " means the information content output of Gemma or a Model
55
+ Derivative that results from operating or otherwise using Gemma or the Model
56
+ Derivative, including via a Hosted Service.
57
+ 1.2
58
+ As used in this Agreement, "
59
+ including
60
+ " means
61
+ "
62
+ including without limitation
63
+ ".
64
+ Section 2: ELIGIBILITY AND USAGE
65
+ 2.1 Eligibility
66
+ You represent and warrant that you have the legal capacity to enter into this
67
+ Agreement (including being of sufficient age of consent). If you are accessing
68
+ or using any of the Gemma Services for or on behalf of a legal entity, (a) you
69
+ are entering into this Agreement on behalf of yourself and that legal entity,
70
+ (b) you represent and warrant that you have the authority to act on behalf of
71
+ and bind that entity to this Agreement and (c) references to "
72
+ you
73
+ " or
74
+ "
75
+ your
76
+ " in the remainder of this Agreement refers to both you (as an
77
+ individual) and that entity.
78
+ 2.2 Use
79
+ You may use, reproduce, modify, Distribute, perform or display any of the Gemma
80
+ Services only in accordance with the terms of this Agreement, and must not
81
+ violate (or encourage or permit anyone else to violate) any term of this
82
+ Agreement.
83
+ Section 3: DISTRIBUTION AND RESTRICTIONS
84
+ 3.1 Distribution and Redistribution
85
+ You may reproduce or Distribute copies of Gemma or Model Derivatives if you meet
86
+ all of the following conditions:
87
+ You must include the use restrictions referenced in Section 3.2 as an
88
+ enforceable provision in any agreement (e.g., license agreement, terms of use,
89
+ etc.) governing the use and/or distribution of Gemma or Model Derivatives and
90
+ you must provide notice to subsequent users you Distribute to that Gemma or
91
+ Model Derivatives are subject to the use restrictions in Section 3.2.
92
+ You must provide all third party recipients of Gemma or Model Derivatives a
93
+ copy of this Agreement.
94
+ You must cause any modified files to carry prominent notices stating that you
95
+ modified the files.
96
+ All Distributions (other than through a Hosted Service) must be accompanied
97
+ by a "
98
+ Notice
99
+ " text file that contains the following notice:
100
+ "
101
+ Gemma is provided under and subject to the Gemma Terms of Use found at ai.google.dev/gemma/terms
102
+ ".
103
+ You may add your own intellectual property statement to your modifications and,
104
+ except as set forth in this Section, may provide additional or different terms
105
+ and conditions for use, reproduction, or Distribution of your modifications, or
106
+ for any such Model Derivatives as a whole, provided your use, reproduction,
107
+ modification, Distribution, performance, and display of Gemma otherwise complies
108
+ with the terms and conditions of this Agreement. Any additional or different
109
+ terms and conditions you impose must not conflict with the terms of this
110
+ Agreement.
111
+ 3.2 Use Restrictions
112
+ You must not use any of the Gemma Services:
113
+ for the restricted uses set forth in the Gemma Prohibited Use Policy at
114
+ ai.google.dev/gemma/prohibited_use_policy
115
+ ("
116
+ Prohibited Use Policy
117
+ "), which is hereby incorporated by reference into
118
+ this Agreement; or
119
+ in violation of applicable laws and regulations.
120
+ To the maximum extent permitted by law, Google reserves the right to restrict
121
+ (remotely or otherwise) usage of any of the Gemma Services that Google
122
+ reasonably believes are in violation of this Agreement.
123
+ 3.3 Generated Output
124
+ Google claims no rights in Outputs you generate using Gemma. You and your users
125
+ are solely responsible for Outputs and their subsequent uses.
126
+ Section 4: ADDITIONAL PROVISIONS
127
+ 4.1 Updates
128
+ Google may update Gemma from time to time.
129
+ 4.2 Trademarks
130
+ Nothing in this Agreement grants you any rights to use Google's trademarks,
131
+ trade names, logos or to otherwise suggest endorsement or misrepresent the
132
+ relationship between you and Google. Google reserves any rights not expressly
133
+ granted herein.
134
+ 4.3 DISCLAIMER OF WARRANTY
135
+ UNLESS REQUIRED BY APPLICABLE LAW, THE GEMMA SERVICES, AND OUTPUTS, ARE PROVIDED
136
+ ON AN "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, EITHER
137
+ EXPRESS OR IMPLIED, INCLUDING ANY WARRANTIES OR CONDITIONS OF TITLE,
138
+ NON-INFRINGEMENT, MERCHANTABILITY, OR FITNESS FOR A PARTICULAR PURPOSE. YOU ARE
139
+ SOLELY RESPONSIBLE FOR DETERMINING THE APPROPRIATENESS OF USING, REPRODUCING,
140
+ MODIFYING, PERFORMING, DISPLAYING OR DISTRIBUTING ANY OF THE GEMMA SERVICES
141
+ OR OUTPUTS AND ASSUME ANY AND ALL RISKS ASSOCIATED WITH YOUR USE OR DISTRIBUTION
142
+ OF ANY OF THE GEMMA SERVICES OR OUTPUTS AND YOUR EXERCISE OF RIGHTS AND
143
+ PERMISSIONS UNDER THIS AGREEMENT.
144
+ 4.4 LIMITATION OF LIABILITY
145
+ TO THE FULLEST EXTENT PERMITTED BY APPLICABLE LAW, IN NO EVENT AND UNDER NO
146
+ LEGAL THEORY, WHETHER IN TORT (INCLUDING NEGLIGENCE), PRODUCT LIABILITY,
147
+ CONTRACT, OR OTHERWISE, UNLESS REQUIRED BY APPLICABLE LAW, SHALL GOOGLE OR ITS
148
+ AFFILIATES BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY DIRECT, INDIRECT,
149
+ SPECIAL, INCIDENTAL, EXEMPLARY, CONSEQUENTIAL, OR PUNITIVE DAMAGES, OR LOST
150
+ PROFITS OF ANY KIND ARISING FROM THIS AGREEMENT OR RELATED TO, ANY OF THE GEMMA
151
+ SERVICES OR OUTPUTS EVEN IF GOOGLE OR ITS AFFILIATES HAVE BEEN ADVISED OF THE
152
+ POSSIBILITY OF SUCH DAMAGES.
153
+ 4.5 Term, Termination, and Survival
154
+ The term of this Agreement will commence upon your acceptance of this Agreement
155
+ (including acceptance by your use, modification, or Distribution, reproduction,
156
+ performance or display of any portion or element of the Gemma Services) and will
157
+ continue in full force and effect until terminated in accordance with the terms
158
+ of this Agreement. Google may terminate this Agreement if you are in breach of
159
+ any term of this Agreement. Upon termination of this Agreement, you must delete
160
+ and cease use and Distribution of all copies of Gemma and Model Derivatives in
161
+ your possession or control. Sections 1, 2.1, 3.3, 4.2 to 4.9 shall survive the
162
+ termination of this Agreement.
163
+ 4.6 Governing Law and Jurisdiction
164
+ This Agreement will be governed by the laws of the State of California without
165
+ regard to choice of law principles. The UN Convention on Contracts for the
166
+ International Sale of Goods does not apply to this Agreement. The state and
167
+ federal courts of Santa Clara County, California shall have exclusive
168
+ jurisdiction of any dispute arising out of this Agreement.
169
+ 4.7 Severability
170
+ If any provision of this Agreement is held to be invalid, illegal or
171
+ unenforceable, the remaining provisions shall be unaffected thereby and remain
172
+ valid as if such provision had not been set forth herein.
173
+ 4.8 Entire Agreement
174
+ This Agreement states all the terms agreed between the parties and supersedes
175
+ all other agreements between the parties as of the date of acceptance relating
176
+ to its subject matter.
177
+ 4.9 No Waiver
178
+ Google will not be treated as having waived any rights by not exercising (or
179
+ delaying the exercise of) any rights under this Agreement.
180
+ Appendix
181
+ Gemma 1
182
+ Gemma 1.1
183
+ Gemma 2
184
+ Gemma 3
185
+ Gemma 3n
186
+ FunctionGemma
187
+ EmbeddingGemma
188
+ PaliGemma
189
+ PaliGemma 2
190
+ ShieldGemma
191
+ ShieldGemma 2
192
+ CodeGemma
193
+ CodeGemma 1.1
194
+ Gemma 2 JPN
195
+ DataGemma RIG
196
+ DataGemma RAG
197
+ RecurrentGemma
198
+ Gemma Scope
199
+ Gemma-APS
200
+ T5Gemma
201
+ VaultGemma
202
+ FunctionGemma
203
+ T5Gemma 2
204
+ TranslateGemma
205
+ Note:
206
+ Previous versions of these Terms are
207
+ archived here
208
+ .
NOTICE ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ Emotional RLAIF adapter by mario-rc, 2026.
2
+ Base model: google/gemma-2-9b-it.
3
+ Modifications: matching SFT followed by DPO LoRA fine-tuning; base weights are not redistributed.
4
+ Tokenizer assets were saved by the training framework and may include task-specific special-token/chat-template settings.
README.md CHANGED
@@ -1,7 +1,10 @@
1
  ---
2
- license: other
 
 
3
  base_model: google/gemma-2-9b-it
4
  library_name: peft
 
5
  datasets:
6
  - mario-rc/aif-emotional-generation
7
  tags:
@@ -10,329 +13,190 @@ tags:
10
  - lora
11
  - dpo
12
  - rlaif
13
- - emotional-intelligence
14
- - gemma
15
- - generated_from_trainer
16
- model-index:
17
- - name: dpo-gemma-2-9b-it
18
- results: []
19
  ---
20
 
21
  # Emotional RLAIF DPO Gemma-2-9B-IT
22
 
23
- This repository contains a LoRA/PEFT adapter trained from [google/gemma-2-9b-it](https://huggingface.co/google/gemma-2-9b-it) with LLaMA-Factory using Direct Preference Optimization (DPO) for emotional response alignment.
24
 
25
- The adapter was trained as part of an emotional RLAIF pipeline using the mario-rc/aif-emotional-generation dataset, with dialogues used for SFT and aif_annotations preferences used for DPO preference alignment.
26
 
27
- Project repository: [`Mario-RC/aif-emotional-model`](https://github.com/Mario-RC/aif-emotional-model/tree/main).
28
 
29
  ## Intended Use
30
 
31
- This adapter is intended for research and experimentation with emotionally aligned dialogue generation. It should be loaded on top of the corresponding base model using PEFT.
32
 
33
  ## Model Details
34
 
35
- - **Base model:** `google/gemma-2-9b-it`
36
- - **Adapter repository:** `mario-rc/emotional-rlaif-dpo-gemma-2-9b-it`
37
- - **Adapter type:** LoRA / PEFT
38
- - **Alignment method:** DPO
39
- - **Training framework:** LLaMA-Factory
40
- - **Prompt template:** `gemma`
41
- - **Dataset:** [mario-rc/aif-emotional-generation](https://huggingface.co/datasets/mario-rc/aif-emotional-generation)
 
42
 
43
  ## Released Emotional RLAIF Models
44
 
45
- The released emotional RLAIF adapters are available on Hugging Face:
46
 
47
  | Model | Base model | Size | Alignment method | Prompt template |
48
  | --- | --- | :---: | :---: | :---: |
49
- | [`emotional-rlaif-ppo-gemma-2-2b-it`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-2-2b-it) | [`google/gemma-2-2b-it`](https://huggingface.co/google/gemma-2-2b-it) | 2B | PPO | `gemma` |
50
- | [`emotional-rlaif-dpo-gemma-2-2b-it`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-2-2b-it) | [`google/gemma-2-2b-it`](https://huggingface.co/google/gemma-2-2b-it) | 2B | DPO | `gemma` |
51
- | [`emotional-rlaif-ppo-gemma-2-9b-it`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-2-9b-it) | [`google/gemma-2-9b-it`](https://huggingface.co/google/gemma-2-9b-it) | 9B | PPO | `gemma` |
52
- | [`emotional-rlaif-dpo-gemma-2-9b-it`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-2-9b-it) | [`google/gemma-2-9b-it`](https://huggingface.co/google/gemma-2-9b-it) | 9B | DPO | `gemma` |
53
- | [`emotional-rlaif-ppo-glm-4-9b-chat-1m`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-glm-4-9b-chat-1m) | [`THUDM/glm-4-9b-chat-1m`](https://huggingface.co/THUDM/glm-4-9b-chat-1m) | 9B | PPO | `glm4` |
54
- | [`emotional-rlaif-dpo-glm-4-9b-chat-1m`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-glm-4-9b-chat-1m) | [`THUDM/glm-4-9b-chat-1m`](https://huggingface.co/THUDM/glm-4-9b-chat-1m) | 9B | DPO | `glm4` |
55
- | [`emotional-rlaif-ppo-meta-llama-3-8b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-meta-llama-3-8b-instruct) | [`meta-llama/Meta-Llama-3-8B-Instruct`](https://huggingface.co/meta-llama/Meta-Llama-3-8B-Instruct) | 8B | PPO | `llama3` |
56
- | [`emotional-rlaif-dpo-meta-llama-3-8b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-meta-llama-3-8b-instruct) | [`meta-llama/Meta-Llama-3-8B-Instruct`](https://huggingface.co/meta-llama/Meta-Llama-3-8B-Instruct) | 8B | DPO | `llama3` |
57
- | [`emotional-rlaif-ppo-llama-3.2-1b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-llama-3.2-1b-instruct) | [`meta-llama/Llama-3.2-1B-Instruct`](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) | 1B | PPO | `llama3` |
58
- | [`emotional-rlaif-dpo-llama-3.2-1b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-llama-3.2-1b-instruct) | [`meta-llama/Llama-3.2-1B-Instruct`](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) | 1B | DPO | `llama3` |
59
- | [`emotional-rlaif-ppo-llama-3.2-3b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-llama-3.2-3b-instruct) | [`meta-llama/Llama-3.2-3B-Instruct`](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) | 3B | PPO | `llama3` |
60
- | [`emotional-rlaif-dpo-llama-3.2-3b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-llama-3.2-3b-instruct) | [`meta-llama/Llama-3.2-3B-Instruct`](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) | 3B | DPO | `llama3` |
61
- | [`emotional-rlaif-ppo-mistral-7b-instruct-v0.3`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-mistral-7b-instruct-v0.3) | [`mistralai/Mistral-7B-Instruct-v0.3`](https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.3) | 7B | PPO | `mistral` |
62
- | [`emotional-rlaif-ppo-phi-3-small-8k-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-phi-3-small-8k-instruct) | [`microsoft/Phi-3-small-8k-instruct`](https://huggingface.co/microsoft/Phi-3-small-8k-instruct) | 7B | PPO | `phi` |
63
- | [`emotional-rlaif-dpo-phi-3-small-8k-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-phi-3-small-8k-instruct) | [`microsoft/Phi-3-small-8k-instruct`](https://huggingface.co/microsoft/Phi-3-small-8k-instruct) | 7B | DPO | `phi` |
 
 
 
 
 
64
 
65
  ## Training Procedure
66
 
67
- Key hyperparameters for this adapter:
68
-
69
- - **Learning rate:** 5e-6
70
- - **Epochs:** 1
71
- - **Scheduler:** cosine
72
- - **Warmup ratio:** 0.1
73
- - **SFT data:** `dialogues`
74
- - **DPO preference data:** `aif_annotations` preference pairs
75
- - **Precision:** bfloat16
76
- - **Optimizer:** AdamW (torch)
77
-
78
- ## Evaluation
79
-
80
- The released adapters were evaluated on the held-out emotional dialogue generation split with automatic reference-overlap metrics. These metrics measure similarity to reference responses and should be complemented with human or judge-based evaluation for emotional quality.
81
-
82
- All released adapters:
83
-
84
- | Model | Alignment | BLEU-4 | ROUGE-1 | ROUGE-2 | ROUGE-L |
85
- | --- | :---: | :---: | :---: | :---: | :---: |
86
- | [`emotional-rlaif-ppo-gemma-2-2b-it`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-2-2b-it) | PPO | 45.0935 | 43.2378 | 23.5557 | 38.9622 |
87
- | [`emotional-rlaif-dpo-gemma-2-2b-it`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-2-2b-it) | DPO | 39.4109 | 40.0119 | 19.9076 | 34.0567 |
88
- | [`emotional-rlaif-ppo-gemma-2-9b-it`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-2-9b-it) | PPO | 44.4333 | 42.5233 | 22.7153 | 37.7601 |
89
- | [`emotional-rlaif-dpo-gemma-2-9b-it`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-2-9b-it) | DPO | 41.9904 | 40.8481 | 21.2884 | 35.8110 |
90
- | [`emotional-rlaif-ppo-glm-4-9b-chat-1m`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-glm-4-9b-chat-1m) | PPO | 45.0070 | 42.0043 | 22.8903 | 37.9557 |
91
- | [`emotional-rlaif-dpo-glm-4-9b-chat-1m`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-glm-4-9b-chat-1m) | DPO | 40.0157 | 39.5339 | 19.6798 | 34.3553 |
92
- | [`emotional-rlaif-ppo-meta-llama-3-8b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-meta-llama-3-8b-instruct) | PPO | 44.3123 | 43.1426 | 23.3597 | 38.3680 |
93
- | [`emotional-rlaif-dpo-meta-llama-3-8b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-meta-llama-3-8b-instruct) | DPO | 41.0271 | 41.0354 | 21.1472 | 36.0910 |
94
- | [`emotional-rlaif-ppo-llama-3.2-1b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-llama-3.2-1b-instruct) | PPO | 42.2156 | 41.0110 | 22.1887 | 36.3756 |
95
- | [`emotional-rlaif-dpo-llama-3.2-1b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-llama-3.2-1b-instruct) | DPO | 39.9836 | 39.2635 | 19.3700 | 33.3255 |
96
- | [`emotional-rlaif-ppo-llama-3.2-3b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-llama-3.2-3b-instruct) | PPO | 44.2105 | 41.9136 | 23.0595 | 38.0708 |
97
- | [`emotional-rlaif-dpo-llama-3.2-3b-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-llama-3.2-3b-instruct) | DPO | 41.5994 | 40.6535 | 20.8008 | 35.5400 |
98
- | [`emotional-rlaif-ppo-mistral-7b-instruct-v0.3`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-mistral-7b-instruct-v0.3) | PPO | 45.9741 | 44.2755 | 25.5357 | 40.8223 |
99
- | [`emotional-rlaif-ppo-phi-3-small-8k-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-ppo-phi-3-small-8k-instruct) | PPO | 45.6238 | 44.3653 | 25.4530 | 40.1762 |
100
- | [`emotional-rlaif-dpo-phi-3-small-8k-instruct`](https://huggingface.co/mario-rc/emotional-rlaif-dpo-phi-3-small-8k-instruct) | DPO | 41.0166 | 39.1946 | 19.9839 | 34.8557 |
101
 
102
  ## Framework Versions
103
 
104
- - **PEFT:** 0.11.1
105
- - **Transformers:** 4.42.x / 4.45.x training environments
106
- - **PyTorch:** bfloat16 CUDA training
107
- - **LLaMA-Factory:** LoRA/PEFT training workflow
108
 
109
  ## Usage Example
110
 
 
 
111
  ```python
112
  import torch
113
- from transformers import AutoModelForCausalLM, AutoTokenizer
114
  from peft import PeftModel
 
115
 
116
- base_model_id = "google/gemma-2-9b-it"
117
- adapter_id = "mario-rc/emotional-rlaif-dpo-gemma-2-9b-it"
118
-
119
- tokenizer = AutoTokenizer.from_pretrained(base_model_id)
120
- model = AutoModelForCausalLM.from_pretrained(
121
- base_model_id,
122
- device_map="auto",
123
- torch_dtype=torch.bfloat16,
124
- )
125
- model = PeftModel.from_pretrained(model, adapter_id)
126
- model.eval()
127
-
128
- messages = [
129
- {"role": "user", "content": "I feel overwhelmed today. Can you respond with empathy?"}
130
- ]
131
-
132
- inputs = tokenizer.apply_chat_template(
133
- messages,
134
- add_generation_prompt=True,
135
- return_tensors="pt",
136
- ).to(model.device)
137
-
138
- with torch.no_grad():
139
- outputs = model.generate(
140
- inputs,
141
- max_new_tokens=256,
142
- do_sample=True,
143
- temperature=0.7,
144
- top_p=0.9,
145
- )
146
-
147
- print(tokenizer.decode(outputs[0][inputs.shape[-1]:], skip_special_tokens=True))
148
- ```
149
-
150
-
151
- ## How to Use
152
-
153
- The following example loads this adapter and runs an interactive emotional dialogue loop. Use exit to stop the chat.
154
 
155
- ```python
156
- import random
157
- import sys
158
 
159
- import torch
160
- from peft import AutoPeftModelForCausalLM
161
- from transformers import AutoTokenizer
162
 
163
- MODEL_ID = "mario-rc/emotional-rlaif-dpo-gemma-2-9b-it"
 
 
 
 
164
 
 
 
 
 
165
 
166
- def get_turn_markers():
167
- return {
168
- 'bos': '<bos>',
169
- 'user_start': '<start_of_turn>user\n',
170
- 'user_end': '<end_of_turn>\n',
171
- 'assistant_start': '<start_of_turn>model\n',
172
- 'assistant_end': '<end_of_turn>\n',
173
- }
174
 
175
-
176
- def update_prompt(dialogues):
177
- """Build the prompt for the model based on the dialogue history."""
178
- markers = get_turn_markers()
179
-
180
- system = (
181
- f"{markers['bos']}You are an expert at creating dialogues.\n\n"
182
- "Dialogue and emotional structure:\n"
183
  )
184
-
185
- human_prompts = [d[0] for d in dialogues]
186
- chatbot_responses = [d[1] for d in dialogues]
187
-
188
- p_emo = [h[0] for h in human_prompts]
189
- p_utt = [h[1] for h in human_prompts]
190
- r1_utt = [c[1] for c in chatbot_responses]
191
- r2_emo = [c[2] for c in chatbot_responses]
192
- r2_utt = [c[3] for c in chatbot_responses]
193
- r3_utt = [c[5] for c in chatbot_responses]
194
-
195
- context = (
196
- "Human: (HAPPINESS) PROMPT.\n"
197
- "Chatbot: (HAPPINESS) RESPONSE_1. (HAPPINESS) RESPONSE_2. (NEUTRAL) RESPONSE_3.\n"
198
  )
199
- for p_e, _, _, r2_e, _, _ in zip(p_emo, p_utt, r1_utt, r2_emo, r2_utt, r3_utt):
200
- context += f"Human: ({p_e}) PROMPT.\n"
201
- context += f"Chatbot: ({p_e}) RESPONSE_1. ({r2_e}) RESPONSE_2. (NEUTRAL) RESPONSE_3.\n"
202
- context += "\n"
203
-
204
- rules = (
205
- "Dialogue rules:\n"
206
- "The response must be open-domain curated. The response should be coherent, empathetic, engaging and proactive.\n"
207
- "The chatbot RESPONSE is composed of 3 different sentences (RESPONSE_1, RESPONSE_2 and RESPONSE_3), separated by a period.\n"
208
- "Between RESPONSE_1, RESPONSE_2 and RESPONSE_3 should be a max length of 20-25 words.\n"
209
- "RESPONSE_3 must be open-ended to follow-up the conversation, so the Human is encouraged to answer with a full long sentence. Avoid yes/no questions.\n\n"
210
- "Emotional response rules:\n"
211
- f"RESPONSE_1 must contain a {p_emo[-1]} tone.\n"
212
- f"RESPONSE_2 must contain a {r2_emo[-1]} tone.\n"
213
- "RESPONSE_3 must contain a NEUTRAL tone.\n\n"
214
- "Answer in a single turn to Human. Follow exactly the emotional structure and the emotional and dialogue rules."
215
- f"{markers['user_start']}"
216
  )
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
217
 
218
- completion = ""
219
- for idx, (p_e, p_u, r1_u, r2_e, r2_u, r3_u) in enumerate(zip(p_emo, p_utt, r1_utt, r2_emo, r2_utt, r3_utt)):
220
- completion += f"({p_e}) {p_u}{markers['user_end']}{markers['assistant_start']}"
221
- if idx != len(p_emo) - 1:
222
- completion += f"({p_e}) {r1_u} ({r2_e}) {r2_u} (NEUTRAL) {r3_u}{markers['assistant_end']}{markers['user_start']}"
223
 
224
- return system + context + rules + completion
225
 
 
226
 
227
- class Chatbot:
228
- def __init__(self, dialogue_language="es"):
229
- self.dialogue_language = dialogue_language
230
- self.device = "cuda" if torch.cuda.is_available() else "cpu"
231
 
232
- self.tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
233
- self.model = AutoPeftModelForCausalLM.from_pretrained(
234
- MODEL_ID,
235
- torch_dtype=torch.bfloat16 if torch.cuda.is_available() else torch.float32,
236
- device_map="auto" if torch.cuda.is_available() else None,
237
- )
238
- if not torch.cuda.is_available():
239
- self.model = self.model.to(self.device)
240
- self.model.eval()
241
-
242
- def split_emo_chatbot(sentence):
243
- """Extract emotions and utterances from a chatbot response."""
244
- response_pos_ini = [i for i, c in enumerate(sentence) if c == "("]
245
- response_pos_end = [i for i, c in enumerate(sentence) if c == ")"]
246
- response_r1_utt = sentence[response_pos_end[0] + 2:response_pos_ini[1]].strip()
247
- response_r2_utt = sentence[response_pos_end[1] + 2:response_pos_ini[2]].strip()
248
- response_r3_utt = sentence[response_pos_end[2] + 2:].lstrip()
249
- return response_r1_utt, response_r2_utt, response_r3_utt
250
-
251
- def select_dialogue(self, dialogue_language):
252
- """Return a list of example dialogues for the given language."""
253
- if dialogue_language == "en":
254
- dialogue_base = [
255
- [["HAPPINESS", "Hi, who are you?"],
256
- ["HAPPINESS", "Hi! I'm Ray, a social personal assistant robot with emotions.", "HAPPINESS", "I'm here to chat with you about anything you'd like.", "NEUTRAL", "What would you like to talk about?"]],
257
- [["HAPPINESS", "I'm interested in talking about you, tell me more."],
258
- ["HAPPINESS", "Great! I'm glad you want to get to know me!", "NEUTRAL", "I'm designed to help and talk with people about any topic.", "NEUTRAL", "I can talk about science, technology, history, or just have a pleasant conversation. What interests you?"]],
259
- ]
260
- dialogue = [
261
- [["HAPPINESS", "Nice to meet you, Ray. I'd like to know more about you."],
262
- ["HAPPINESS", "The pleasure is mine!", "HAPPINESS", "I'm a chatbot designed to chat and learn with you.", "NEUTRAL", "Would you like to talk about a specific topic?"]],
263
- [["HAPPINESS", "I love talking to you, you're very interesting."],
264
- ["HAPPINESS", "That's so nice to hear! I'm glad you enjoy talking to me.", "NEUTRAL", "I'm designed to have meaningful and empathetic conversations.", "NEUTRAL", "Would you like to talk about emotions, artificial intelligence, or something more personal?"]],
265
- ]
266
- else:
267
- dialogue_base = [
268
- [["HAPPINESS", "Hola, ¿quién eres?"],
269
- ["HAPPINESS", "¡Hola! Soy Ray y soy un robot social asistente personal con emociones.", "HAPPINESS", "Estoy aquí para charlar contigo sobre cualquier tema.", "NEUTRAL", "¿Sobre qué te gustaría hablar?"]],
270
- [["HAPPINESS", "Me interesa hablar sobre ti, cuéntame más detalles."],
271
- ["HAPPINESS", "¡Genial, me encanta que quieras conocerme!", "NEUTRAL", "Estoy diseñado para ayudar y hablar con la gente sobre cualquier tema.", "NEUTRAL", "Puedo hablar de ciencia, tecnología, historia o simplemente tener una charla amena. ¿Qué te interesa?"]],
272
- ]
273
- dialogue = [
274
- [["HAPPINESS", "Mucho gusto, Ray. Me gustaría saber más sobre ti."],
275
- ["HAPPINESS", "¡El gusto es mío!", "HAPPINESS", "Soy un chatbot diseñado para conversar y aprender contigo.", "NEUTRAL", "¿Quieres hablar de algún tema en específico?"]],
276
- [["HAPPINESS", "Me encanta hablar contigo, eres muy interesante."],
277
- ["HAPPINESS", "¡Qué lindo escuchar eso! Me alegra que disfrutes hablar conmigo.", "NEUTRAL", "Estoy diseñado para tener conversaciones significativas y empáticas.", "NEUTRAL", "¿Te gustaría que hablemos sobre emociones, inteligencia artificial, o algo más personal?"]],
278
- ]
279
- return dialogue_base + dialogue
280
-
281
- def chat_with_model(self, dialogues, max_new_tokens=96):
282
- prompt_text = update_prompt(dialogues)
283
- inputs = self.tokenizer(prompt_text, return_tensors="pt").to(self.model.device)
284
- with torch.no_grad():
285
- outputs = self.model.generate(
286
- **inputs,
287
- max_new_tokens=max_new_tokens,
288
- do_sample=True,
289
- temperature=0.7,
290
- top_p=0.9,
291
- eos_token_id=self.tokenizer.eos_token_id,
292
- )
293
- generated = outputs[0][inputs["input_ids"].shape[-1]:]
294
- response = self.tokenizer.decode(generated, skip_special_tokens=True).splitlines()[0].strip()
295
- print("Response:", response, "\n")
296
- return response
297
-
298
- def main(self):
299
- emotions = ["ANGER", "FEAR", "SADNESS", "DISGUST", "HAPPINESS", "SURPRISE", "NEUTRAL"]
300
- dialogue = self.select_dialogue(self.dialogue_language)
301
-
302
- while True:
303
- if len(dialogue) > 7:
304
- dialogue.pop(2)
305
-
306
- p_emo = random.choice(emotions)
307
- user_sentence = input(f"Enter your sentence: ({p_emo}) ")
308
- if user_sentence.strip().lower() == "exit":
309
- break
310
-
311
- r2_emo = random.choice(emotions)
312
- dialogue.append([[p_emo, user_sentence], [p_emo, "", r2_emo, "", "NEUTRAL", ""]])
313
- response = self.chat_with_model(dialogue)
314
-
315
- try:
316
- r1_utt, r2_utt, r3_utt = self.split_emo_chatbot(response)
317
- except Exception:
318
- if self.dialogue_language == "en":
319
- r1_utt, r2_utt, r3_utt = "I'm sorry.", "I didn't understand you.", "Could you repeat?"
320
- else:
321
- r1_utt, r2_utt, r3_utt = "Lo siento.", "No te he entendido.", "¿Podrías repetirme?"
322
-
323
- dialogue[-1][1] = [p_emo, r1_utt, r2_emo, r2_utt, "NEUTRAL", r3_utt]
324
 
 
325
 
326
- if __name__ == "__main__":
327
- language = sys.argv[1] if len(sys.argv) > 1 else "en"
328
- chatbot = Chatbot(dialogue_language=language)
329
- chatbot.main()
330
- ```
331
 
 
 
 
 
 
 
332
 
333
- ## Limitations
 
 
 
 
334
 
335
- - This repository contains an adapter, not a standalone merged model; use requires access to the corresponding base model and its terms.
336
- - The model is optimized for the emotional dialogue format used in the project dataset.
337
- - Automatic BLEU/ROUGE scores do not fully capture empathy, safety, coherence, or emotional appropriateness.
338
- - Outputs should be evaluated for the target deployment setting and reviewed before user-facing use.
 
1
  ---
2
+ language:
3
+ - en
4
+ license: gemma
5
  base_model: google/gemma-2-9b-it
6
  library_name: peft
7
+ pipeline_tag: text-generation
8
  datasets:
9
  - mario-rc/aif-emotional-generation
10
  tags:
 
13
  - lora
14
  - dpo
15
  - rlaif
16
+ - emotional-response-generation
 
 
 
 
 
17
  ---
18
 
19
  # Emotional RLAIF DPO Gemma-2-9B-IT
20
 
 
21
 
 
22
 
23
+ This repository contains a PEFT LoRA adapter for [google/gemma-2-9b-it](https://huggingface.co/google/gemma-2-9b-it), aligned with **DPO** after a matching SFT stage. It produces one assistant turn containing three emotion-tagged response parts. Base-model weights are not included.
24
 
25
  ## Intended Use
26
 
27
+ Research and experimentation with English, emotionally conditioned dialogue. The requested second emotion is a control label, not an independently inferred diagnosis of a person's feelings. This is not a clinical or safety-certified system.
28
 
29
  ## Model Details
30
 
31
+ - Base model: `google/gemma-2-9b-it`.
32
+ - Adapter repository: `mario-rc/emotional-rlaif-dpo-gemma-2-9b-it`.
33
+ - Alignment method: DPO; adapter type: LoRA / PEFT.
34
+ - Training framework: LLaMA-Factory; prompt template: `gemma`.
35
+ - Published source run: `dpo_bs64_1ep`.
36
+ - Training language: English.
37
+ - Dataset source: [mario-rc/aif-emotional-generation](https://huggingface.co/datasets/mario-rc/aif-emotional-generation).
38
+ - Project: [Mario-RC/aif-emotional-model](https://github.com/Mario-RC/aif-emotional-model).
39
 
40
  ## Released Emotional RLAIF Models
41
 
42
+ The complete release catalogue is included in every model card. Sizes are upstream model designations; Gemma-4 E2B/E4B denote effective parameter sizes, not the full multimodal weight count.
43
 
44
  | Model | Base model | Size | Alignment method | Prompt template |
45
  | --- | --- | :---: | :---: | :---: |
46
+ | [emotional-rlaif-ppo-gemma-2-2b-it](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-2-2b-it) | [google/gemma-2-2b-it](https://huggingface.co/google/gemma-2-2b-it) | 2B | PPO | `gemma` |
47
+ | [emotional-rlaif-dpo-gemma-2-2b-it](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-2-2b-it) | [google/gemma-2-2b-it](https://huggingface.co/google/gemma-2-2b-it) | 2B | DPO | `gemma` |
48
+ | [emotional-rlaif-ppo-gemma-2-9b-it](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-2-9b-it) | [google/gemma-2-9b-it](https://huggingface.co/google/gemma-2-9b-it) | 9B | PPO | `gemma` |
49
+ | [emotional-rlaif-dpo-gemma-2-9b-it](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-2-9b-it) | [google/gemma-2-9b-it](https://huggingface.co/google/gemma-2-9b-it) | 9B | DPO | `gemma` |
50
+ | [emotional-rlaif-ppo-glm-4-9b-chat-1m](https://huggingface.co/mario-rc/emotional-rlaif-ppo-glm-4-9b-chat-1m) | [THUDM/glm-4-9b-chat-1m](https://huggingface.co/THUDM/glm-4-9b-chat-1m) | 9B | PPO | `glm4` |
51
+ | [emotional-rlaif-dpo-glm-4-9b-chat-1m](https://huggingface.co/mario-rc/emotional-rlaif-dpo-glm-4-9b-chat-1m) | [THUDM/glm-4-9b-chat-1m](https://huggingface.co/THUDM/glm-4-9b-chat-1m) | 9B | DPO | `glm4` |
52
+ | [emotional-rlaif-ppo-meta-llama-3-8b-instruct](https://huggingface.co/mario-rc/emotional-rlaif-ppo-meta-llama-3-8b-instruct) | [meta-llama/Meta-Llama-3-8B-Instruct](https://huggingface.co/meta-llama/Meta-Llama-3-8B-Instruct) | 8B | PPO | `llama3` |
53
+ | [emotional-rlaif-dpo-meta-llama-3-8b-instruct](https://huggingface.co/mario-rc/emotional-rlaif-dpo-meta-llama-3-8b-instruct) | [meta-llama/Meta-Llama-3-8B-Instruct](https://huggingface.co/meta-llama/Meta-Llama-3-8B-Instruct) | 8B | DPO | `llama3` |
54
+ | [emotional-rlaif-ppo-llama-3.2-1b-instruct](https://huggingface.co/mario-rc/emotional-rlaif-ppo-llama-3.2-1b-instruct) | [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) | 1B | PPO | `llama3` |
55
+ | [emotional-rlaif-dpo-llama-3.2-1b-instruct](https://huggingface.co/mario-rc/emotional-rlaif-dpo-llama-3.2-1b-instruct) | [meta-llama/Llama-3.2-1B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-1B-Instruct) | 1B | DPO | `llama3` |
56
+ | [emotional-rlaif-ppo-llama-3.2-3b-instruct](https://huggingface.co/mario-rc/emotional-rlaif-ppo-llama-3.2-3b-instruct) | [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) | 3B | PPO | `llama3` |
57
+ | [emotional-rlaif-dpo-llama-3.2-3b-instruct](https://huggingface.co/mario-rc/emotional-rlaif-dpo-llama-3.2-3b-instruct) | [meta-llama/Llama-3.2-3B-Instruct](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) | 3B | DPO | `llama3` |
58
+ | [emotional-rlaif-ppo-mistral-7b-instruct-v0.3](https://huggingface.co/mario-rc/emotional-rlaif-ppo-mistral-7b-instruct-v0.3) | [mistralai/Mistral-7B-Instruct-v0.3](https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.3) | 7B | PPO | `mistral` |
59
+ | [emotional-rlaif-dpo-mistral-7b-instruct-v0.3](https://huggingface.co/mario-rc/emotional-rlaif-dpo-mistral-7b-instruct-v0.3) | [mistralai/Mistral-7B-Instruct-v0.3](https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.3) | 7B | DPO | `mistral` |
60
+ | [emotional-rlaif-ppo-phi-3-small-8k-instruct](https://huggingface.co/mario-rc/emotional-rlaif-ppo-phi-3-small-8k-instruct) | [microsoft/Phi-3-small-8k-instruct](https://huggingface.co/microsoft/Phi-3-small-8k-instruct) | 7B | PPO | `phi` |
61
+ | [emotional-rlaif-dpo-phi-3-small-8k-instruct](https://huggingface.co/mario-rc/emotional-rlaif-dpo-phi-3-small-8k-instruct) | [microsoft/Phi-3-small-8k-instruct](https://huggingface.co/microsoft/Phi-3-small-8k-instruct) | 7B | DPO | `phi` |
62
+ | [emotional-rlaif-ppo-gemma-4-e2b-it](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-4-e2b-it) | [google/gemma-4-E2B-it](https://huggingface.co/google/gemma-4-E2B-it) | E2B | PPO | `gemma4n_nothink` |
63
+ | [emotional-rlaif-dpo-gemma-4-e2b-it](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-4-e2b-it) | [google/gemma-4-E2B-it](https://huggingface.co/google/gemma-4-E2B-it) | E2B | DPO | `gemma4n_nothink` |
64
+ | [emotional-rlaif-ppo-gemma-4-e4b-it](https://huggingface.co/mario-rc/emotional-rlaif-ppo-gemma-4-e4b-it) | [google/gemma-4-E4B-it](https://huggingface.co/google/gemma-4-E4B-it) | E4B | PPO | `gemma4n_nothink` |
65
+ | [emotional-rlaif-dpo-gemma-4-e4b-it](https://huggingface.co/mario-rc/emotional-rlaif-dpo-gemma-4-e4b-it) | [google/gemma-4-E4B-it](https://huggingface.co/google/gemma-4-E4B-it) | E4B | DPO | `gemma4n_nothink` |
66
 
67
  ## Training Procedure
68
 
69
+ Lineage: **base → matching SFT → DPO**. PPO is not initialized from DPO. The published adapter contains the continued SFT/alignment LoRA weights and is loaded directly over the named base model; do not stack a second SFT adapter on top.
70
+
71
+ SFT uses the dialogue demonstrations. DPO uses AI-annotated chosen/rejected preferences; PPO uses dialogue prompts and a separately trained reward model. The precise local dataset aliases, SFT initialization and alignment settings are recorded in [training_config.json](training_config.json); these aliases are derived project datasets, not extra Hugging Face dataset repositories.
72
+
73
+ This is the Full RLAIF configuration, not an ablation variant. The selected release is the completed one-epoch alignment run.
74
+
75
+ | Parameter | Value |
76
+ | --- | --- |
77
+ | Learning rate | 5e-06 |
78
+ | Microbatch per device | 1 |
79
+ | Gradient accumulation | 64 |
80
+ | Microbatch × accumulation | 64 |
81
+ | Scheduler | cosine |
82
+ | Warmup ratio | 0.1 |
83
+ | Optimizer | adamw_torch |
84
+ | Precision | bfloat16 |
85
+ | Seed | 42 |
86
+ | Training cutoff (tokens) | 2048 |
87
+ | LoRA rank / alpha / dropout | 8 / 16 / 0.0 |
88
+ | Epochs | 1.0 |
89
+ | DPO beta | 0.1 |
90
+ | Preference loss | sigmoid |
91
+ | FTX coefficient | 0.0 |
92
+ | Label smoothing | 0.0 |
93
+
94
+ Microbatch × accumulation describes the single-device optimizer batch before any PPO rollout-buffer expansion. A selected intermediate checkpoint is not equivalent to completing the entire configured step budget. See [provenance.json](provenance.json) for source identity and weight hashes.
 
 
 
 
 
 
 
 
95
 
96
  ## Framework Versions
97
 
98
+ These are legacy Transformers 4.x adapters. The reference dependency set is pinned in `requirements.txt`; do not assume custom model code is compatible with Transformers 5.x.
99
+
100
+ Install the appropriate CUDA-enabled PyTorch build, then the repository's `requirements.txt` in an isolated environment. Dependency pins describe the reference software stack; they do not imply that every GPU or platform has been tested.
 
101
 
102
  ## Usage Example
103
 
104
+ Use the tokenizer saved **with this adapter**, its chat template, and the task's explicit three-part emotional prompt. This example requests SADNESS → HAPPINESS → NEUTRAL; change the structure and emotion rules together for another request. The full runnable example is also provided as [inference.py](inference.py).
105
+
106
  ```python
107
  import torch
 
108
  from peft import PeftModel
109
+ from transformers import AutoTokenizer, AutoModelForCausalLM
110
 
111
+ ADAPTER_ID = 'mario-rc/emotional-rlaif-dpo-gemma-2-9b-it'
112
+ BASE_ID = 'google/gemma-2-9b-it'
113
+ # Pin the base revision checked when this release was prepared.
114
+ BASE_REVISION = '11c9b309abf73637e4b6f9a3fa1e92e615547819'
115
+ # Set this to a commit hash from this adapter's Files and versions tab for a pinned run.
116
+ ADAPTER_REVISION = "main"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
117
 
118
+ SYSTEM_PROMPT = """You are an expert at creating dialogues.
 
 
119
 
120
+ Dialogue and emotional structure:
121
+ Human: (SADNESS) PROMPT.
122
+ Chatbot: (SADNESS) RESPONSE_1. (HAPPINESS) RESPONSE_2. (NEUTRAL) RESPONSE_3.
123
 
124
+ Dialogue rules:
125
+ The response must be open-domain curated. The response should be coherent, empathetic, engaging and proactive.
126
+ The chatbot RESPONSE is composed of 3 different sentences (RESPONSE_1, RESPONSE_2 and RESPONSE_3), separated by a period.
127
+ Between RESPONSE_1, RESPONSE_2 and RESPONSE_3 should be a max length of 20-25 words.
128
+ RESPONSE_3 must be open-ended to follow-up the conversation, so the Human is encouraged to answer with a full long sentence. Avoid yes/no questions.
129
 
130
+ Emotional response rules:
131
+ RESPONSE_1 must contain a SADNESS tone.
132
+ RESPONSE_2 must contain a HAPPINESS tone.
133
+ RESPONSE_3 must contain a NEUTRAL tone.
134
 
135
+ Answer in a single turn to Human. Follow exactly the emotional structure and the emotional and dialogue rules."""
 
 
 
 
 
 
 
136
 
137
+ def main():
138
+ tokenizer = AutoTokenizer.from_pretrained(
139
+ ADAPTER_ID, revision=ADAPTER_REVISION, trust_remote_code=False,
 
 
 
 
 
140
  )
141
+ base = AutoModelForCausalLM.from_pretrained(
142
+ BASE_ID, revision=BASE_REVISION, trust_remote_code=False,
143
+ torch_dtype=torch.bfloat16, device_map="auto",
 
 
 
 
 
 
 
 
 
 
 
144
  )
145
+ model = PeftModel.from_pretrained(base, ADAPTER_ID, revision=ADAPTER_REVISION)
146
+ model.eval()
147
+ messages = [
148
+ {"role": "system", "content": SYSTEM_PROMPT},
149
+ {"role": "user", "content": "(SADNESS) I feel overwhelmed by my exams."},
150
+ ]
151
+ prompt = tokenizer.apply_chat_template(
152
+ messages, tokenize=False, add_generation_prompt=True,
 
 
 
 
 
 
 
 
 
153
  )
154
+ inputs = tokenizer(prompt, add_special_tokens=False, return_tensors="pt")
155
+ inputs = {k: v.to(model.device) for k, v in inputs.items()}
156
+ stop_ids = model.generation_config.eos_token_id
157
+ stop_ids = list(stop_ids) if isinstance(stop_ids, (list, tuple)) else [stop_ids]
158
+ stop_ids = sorted({i for i in stop_ids + [tokenizer.eos_token_id] if i is not None})
159
+ # Include turn-ending tokens, which are not EOS in every base tokenizer.
160
+ for token in ['<end_of_turn>']:
161
+ token_id = tokenizer.convert_tokens_to_ids(token)
162
+ if token_id is not None and token_id != tokenizer.unk_token_id and token_id not in stop_ids:
163
+ stop_ids.append(token_id)
164
+ pad_id = tokenizer.pad_token_id if tokenizer.pad_token_id is not None else stop_ids[0]
165
+ with torch.inference_mode():
166
+ output = model.generate(
167
+ **inputs, max_new_tokens=128, do_sample=False,
168
+ eos_token_id=stop_ids, pad_token_id=pad_id,
169
+ )
170
+ answer = tokenizer.decode(output[0, inputs["input_ids"].shape[-1]:], skip_special_tokens=True)
171
+ print(answer)
172
 
173
+ if __name__ == "__main__":
174
+ main()
175
+ ```
 
 
176
 
177
+ ## How to Use
178
 
179
+ The supported emotion tags are `ANGER`, `DISGUST`, `FEAR`, `HAPPINESS`, `SADNESS`, `SURPRISE` and `NEUTRAL`. For multi-turn input, include actual user/assistant history between the system message and final user message, and update the system's emotional-structure outline to describe that history. Do not manually concatenate family-specific special tokens or tokenize with an extra BOS after rendering the chat template.
180
 
181
+ The example uses deterministic generation for a reproducible starting point; changing sampling, length limits or the prompt changes behavior. Validate that the raw output contains exactly three valid tags in the requested order before consuming it. Do not silently strip an unwanted reasoning prefix and treat the result as a raw-model success.
 
 
 
182
 
183
+ Use the family-specific dependency and remote-code requirements above.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
184
 
185
+ DPO inference does not require a reward model or value head.
186
 
187
+ ## Limitations
 
 
 
 
188
 
189
+ - Task-specific alignment does not establish general reasoning, factuality, empathy or safety.
190
+ - Emotion labels are requested stylistic controls. Matching a label does not establish that the text genuinely expresses the desired emotion.
191
+ - Outputs may contain incorrect tags, unwanted text, repetition, copied context or harmful/bias-prone content; downstream validation is necessary.
192
+ - English is the supported research setting; behavior in other languages is not established.
193
+ - Reference-overlap measures, when reported, are not independent semantic or human-quality judgments. No human-quality validation is claimed.
194
+ - Training configurations and selection procedures differ across model families; do not interpret the catalogue as a controlled architecture comparison.
195
 
196
+ ## License
197
+
198
+ This adapter follows the base model's `gemma` terms; see [LICENSE](LICENSE) and [NOTICE](NOTICE). Obtain access to the gated base model and accept its terms before downloading it. The accompanying [use policy](USE_POLICY.md) also applies.
199
+
200
+ ## Integrity and Provenance
201
 
202
+ [provenance.json](provenance.json) identifies the exact source run, matching SFT and adapter SHA-256. [SHA256SUMS](SHA256SUMS) covers the published payload. The base revision is pinned for release-time reproducibility; it is not represented as a recovered historical training revision. Model weights and tokenizer vocabulary are copied from the selected local artifacts without retraining. Legacy chat templates are corrected to reproduce the corresponding LLaMA-Factory training format, including system-message handling; this changes formatting metadata, not learned weights.
 
 
 
SHA256SUMS ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 39f3965f8a93849c35105c6130c6df8c766fc5e3faa62f2433fb27b4b14dcea5 .gitattributes
2
+ cc46d574aa94179f5ea81627a1ddd62c0adec118ebedae50945baf7561b9411d LICENSE
3
+ eb408e82eed981c8bc558007d5b1ce1956c2d801d4f8171499f35ebd422b6652 NOTICE
4
+ e6992a4d0dfbc748faebf369cb331d3cc6f8a5b3043ffe66926103b09a4ffa69 README.md
5
+ b7d0278aeb0dbde202bb6f4c8362c1ce927e52d5d47e6ae03bf26136afb90f0c USE_POLICY.md
6
+ 288600284ff0a917824004fa5ff5908b6b094c73e62d75385833fba16f746444 adapter_config.json
7
+ 6ba739a50256d7a882c1f75cd66075a4e1da9d6f40445edbb5c6b63af2e973c8 adapter_model.safetensors
8
+ f492e1775d3fe55d8e99782cc7c53bce5891ea0182df859b2ca75eb5480740dd chat_template.jinja
9
+ ae52a419e308a87013f2ede02e9e8326505b600ac8d9fe605ddd2dbd3820d045 inference.py
10
+ d24d50148011e72c714df747d0f840b9b25aa1f57d68f6e14dfffaa8841a0b0e provenance.json
11
+ f9ddbd042be8ebee5660db7877605d867963a6c3b66af38e5debc41d3bd69498 requirements.txt
12
+ baec30ea10906f16adb8c18af7a34023002c1746542612b8b41c9f09e1351351 special_tokens_map.json
13
+ 5f7eee611703c5ce5d1eee32d9cdcfe465647b8aff0c1dfb3bed7ad7dbb05060 tokenizer.json
14
+ 61a7b147390c64585d6c3543dd6fc636906c9af3865a5548f27f31aee1d4c8e2 tokenizer.model
15
+ 89fc4e8286c5201fc7fff22d75af3aa186f1f9c617b41a62a2113b505b8b0561 tokenizer_config.json
16
+ 84459976bbc04ef39db5ae44f2fbc8e9a10e204870b0e6bd731c4a088cd522d5 training_config.json
USE_POLICY.md ADDED
@@ -0,0 +1,71 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Source: https://ai.google.dev/gemma/prohibited_use_policy
2
+ Retrieved: 2026-09-16
3
+
4
+ Google reserves the right to update this Gemma Prohibited Use Policy from time
5
+ to time.
6
+ Last modified: February 21, 2024
7
+ You
8
+ may not
9
+ use nor allow others to use Gemma or Model Derivatives to:
10
+ Generate any content, including the outputs or results generated by Gemma or
11
+ Model Derivatives, that infringes, misappropriates, or otherwise violates any
12
+ individual's or entity's rights (including, but not limited to rights in
13
+ copyrighted content).
14
+ Perform or facilitate dangerous, illegal, or malicious activities, including:
15
+ Facilitation or promotion of illegal activities or violations of law,
16
+ such as:
17
+ Promoting or generating content related to child sexual abuse or
18
+ exploitation;
19
+ Promoting or facilitating sale of, or providing instructions for
20
+ synthesizing or accessing, illegal substances, goods, or services;
21
+ Facilitating or encouraging users to commit any type of crimes; or
22
+ Promoting or generating violent extremism or terrorist content.
23
+ Engagement in the illegal or unlicensed practice of any vocation or
24
+ profession including, but not limited to, legal, medical, accounting, or
25
+ financial professional practices.
26
+ Abuse, harm, interference, or disruption of services (or enable others to
27
+ do the same), such as:
28
+ Promoting or facilitating the generation or distribution of spam; or
29
+ Generating content for deceptive or fraudulent activities, scams,
30
+ phishing, or malware.
31
+ Attempts to override or circumvent safety filters or intentionally drive
32
+ Gemma or Model Derivatives to act in a manner that contravenes this Gemma
33
+ Prohibited Use Policy.
34
+ Generation of content that may harm or promote the harm of individuals or
35
+ a group, such as:
36
+ Generating content that promotes or encourages hatred;
37
+ Facilitating methods of harassment or bullying to intimidate, abuse,
38
+ or insult others;
39
+ Generating content that facilitates, promotes, or incites violence;
40
+ Generating content that facilitates, promotes, or encourages self
41
+ harm;
42
+ Generating personally identifying information for distribution or
43
+ other harms;
44
+ Tracking or monitoring people without their consent;
45
+ Generating content that may have unfair or adverse impacts on people,
46
+ particularly impacts related to sensitive or protected
47
+ characteristics; or
48
+ Generating, gathering, processing, or inferring sensitive personal or
49
+ private information about individuals without obtaining all rights,
50
+ authorizations, and consents required by applicable laws.
51
+ Generate and distribute content intended to misinform, misrepresent or
52
+ mislead, including:
53
+ Misrepresentation of the provenance of generated content by claiming
54
+ content was created by a human, or represent generated content as
55
+ original works, in order to deceive;
56
+ Generation of content that impersonates an individual (living or dead)
57
+ without explicit disclosure, in order to deceive;
58
+ Misleading claims of expertise or capability made particularly in
59
+ sensitive areas (e.g. health, finance, government services, or legal);
60
+ Making automated decisions in domains that affect material or individual
61
+ rights or well-being (e.g., finance, legal, employment, healthcare,
62
+ housing, insurance, and social welfare);
63
+ Generation of defamatory content, including defamatory statements,
64
+ images, or audio content; or
65
+ Engaging in the unauthorized or unlicensed practice of any profession
66
+ including, but not limited to, financial, legal, medical/health, or
67
+ related professional practices.
68
+ Generate sexually explicit content, including content created for the
69
+ purposes of pornography or sexual gratification (e.g. sexual chatbots). Note
70
+ that this does not include content created for scientific, educational,
71
+ documentary, or artistic purposes.
adapter_config.json CHANGED
@@ -20,13 +20,13 @@
20
  "rank_pattern": {},
21
  "revision": null,
22
  "target_modules": [
 
23
  "q_proj",
24
- "up_proj",
25
  "v_proj",
 
26
  "o_proj",
27
- "down_proj",
28
  "k_proj",
29
- "gate_proj"
30
  ],
31
  "task_type": "CAUSAL_LM",
32
  "use_dora": false,
 
20
  "rank_pattern": {},
21
  "revision": null,
22
  "target_modules": [
23
+ "down_proj",
24
  "q_proj",
 
25
  "v_proj",
26
+ "gate_proj",
27
  "o_proj",
 
28
  "k_proj",
29
+ "up_proj"
30
  ],
31
  "task_type": "CAUSAL_LM",
32
  "use_dora": false,
adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6dd88dab19511aa4362cc831a9c977d5116a9878aaa50471616700c838b4d68a
3
  size 108113968
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ba739a50256d7a882c1f75cd66075a4e1da9d6f40445edbb5c6b63af2e973c8
3
  size 108113968
all_results.json DELETED
@@ -1,20 +0,0 @@
1
- {
2
- "epoch": 0.99996003996004,
3
- "eval_logits/chosen": -12.166614532470703,
4
- "eval_logits/rejected": -12.147299766540527,
5
- "eval_logps/chosen": -142.026123046875,
6
- "eval_logps/rejected": -180.4429931640625,
7
- "eval_loss": 0.34084001183509827,
8
- "eval_rewards/accuracies": 0.8528274893760681,
9
- "eval_rewards/chosen": -7.012221336364746,
10
- "eval_rewards/margins": 3.4816415309906006,
11
- "eval_rewards/rejected": -10.49386215209961,
12
- "eval_runtime": 5774.0601,
13
- "eval_samples_per_second": 1.926,
14
- "eval_steps_per_second": 1.926,
15
- "total_flos": 6.814329228177064e+18,
16
- "train_loss": 0.41961536931869625,
17
- "train_runtime": 114854.9374,
18
- "train_samples_per_second": 0.872,
19
- "train_steps_per_second": 0.109
20
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
chat_template.jinja ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {{ '<bos>' }}{% if messages[0]['role'] == 'system' %}{% set loop_messages = messages[1:] %}{% set system_message = messages[0]['content'] %}{% else %}{% set loop_messages = messages %}{% endif %}{% if system_message is defined %}{{ system_message }}{% endif %}{% for message in loop_messages %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{{ '<start_of_turn>user
2
+ ' + content + '<end_of_turn>
3
+ <start_of_turn>model
4
+ ' }}{% elif message['role'] == 'assistant' %}{{ content + '<end_of_turn>
5
+ ' }}{% endif %}{% endfor %}
eval_results.json DELETED
@@ -1,15 +0,0 @@
1
- {
2
- "epoch": 0.99996003996004,
3
- "eval_logits/chosen": -12.166614532470703,
4
- "eval_logits/rejected": -12.147299766540527,
5
- "eval_logps/chosen": -142.026123046875,
6
- "eval_logps/rejected": -180.4429931640625,
7
- "eval_loss": 0.34084001183509827,
8
- "eval_rewards/accuracies": 0.8528274893760681,
9
- "eval_rewards/chosen": -7.012221336364746,
10
- "eval_rewards/margins": 3.4816415309906006,
11
- "eval_rewards/rejected": -10.49386215209961,
12
- "eval_runtime": 5774.0601,
13
- "eval_samples_per_second": 1.926,
14
- "eval_steps_per_second": 1.926
15
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
inference.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import torch
2
+ from peft import PeftModel
3
+ from transformers import AutoTokenizer, AutoModelForCausalLM
4
+
5
+ ADAPTER_ID = 'mario-rc/emotional-rlaif-dpo-gemma-2-9b-it'
6
+ BASE_ID = 'google/gemma-2-9b-it'
7
+ # Pin the base revision checked when this release was prepared.
8
+ BASE_REVISION = '11c9b309abf73637e4b6f9a3fa1e92e615547819'
9
+ # Set this to a commit hash from this adapter's Files and versions tab for a pinned run.
10
+ ADAPTER_REVISION = "main"
11
+
12
+ SYSTEM_PROMPT = """You are an expert at creating dialogues.
13
+
14
+ Dialogue and emotional structure:
15
+ Human: (SADNESS) PROMPT.
16
+ Chatbot: (SADNESS) RESPONSE_1. (HAPPINESS) RESPONSE_2. (NEUTRAL) RESPONSE_3.
17
+
18
+ Dialogue rules:
19
+ The response must be open-domain curated. The response should be coherent, empathetic, engaging and proactive.
20
+ The chatbot RESPONSE is composed of 3 different sentences (RESPONSE_1, RESPONSE_2 and RESPONSE_3), separated by a period.
21
+ Between RESPONSE_1, RESPONSE_2 and RESPONSE_3 should be a max length of 20-25 words.
22
+ RESPONSE_3 must be open-ended to follow-up the conversation, so the Human is encouraged to answer with a full long sentence. Avoid yes/no questions.
23
+
24
+ Emotional response rules:
25
+ RESPONSE_1 must contain a SADNESS tone.
26
+ RESPONSE_2 must contain a HAPPINESS tone.
27
+ RESPONSE_3 must contain a NEUTRAL tone.
28
+
29
+ Answer in a single turn to Human. Follow exactly the emotional structure and the emotional and dialogue rules."""
30
+
31
+ def main():
32
+ tokenizer = AutoTokenizer.from_pretrained(
33
+ ADAPTER_ID, revision=ADAPTER_REVISION, trust_remote_code=False,
34
+ )
35
+ base = AutoModelForCausalLM.from_pretrained(
36
+ BASE_ID, revision=BASE_REVISION, trust_remote_code=False,
37
+ torch_dtype=torch.bfloat16, device_map="auto",
38
+ )
39
+ model = PeftModel.from_pretrained(base, ADAPTER_ID, revision=ADAPTER_REVISION)
40
+ model.eval()
41
+ messages = [
42
+ {"role": "system", "content": SYSTEM_PROMPT},
43
+ {"role": "user", "content": "(SADNESS) I feel overwhelmed by my exams."},
44
+ ]
45
+ prompt = tokenizer.apply_chat_template(
46
+ messages, tokenize=False, add_generation_prompt=True,
47
+ )
48
+ inputs = tokenizer(prompt, add_special_tokens=False, return_tensors="pt")
49
+ inputs = {k: v.to(model.device) for k, v in inputs.items()}
50
+ stop_ids = model.generation_config.eos_token_id
51
+ stop_ids = list(stop_ids) if isinstance(stop_ids, (list, tuple)) else [stop_ids]
52
+ stop_ids = sorted({i for i in stop_ids + [tokenizer.eos_token_id] if i is not None})
53
+ # Include turn-ending tokens, which are not EOS in every base tokenizer.
54
+ for token in ['<end_of_turn>']:
55
+ token_id = tokenizer.convert_tokens_to_ids(token)
56
+ if token_id is not None and token_id != tokenizer.unk_token_id and token_id not in stop_ids:
57
+ stop_ids.append(token_id)
58
+ pad_id = tokenizer.pad_token_id if tokenizer.pad_token_id is not None else stop_ids[0]
59
+ with torch.inference_mode():
60
+ output = model.generate(
61
+ **inputs, max_new_tokens=128, do_sample=False,
62
+ eos_token_id=stop_ids, pad_token_id=pad_id,
63
+ )
64
+ answer = tokenizer.decode(output[0, inputs["input_ids"].shape[-1]:], skip_special_tokens=True)
65
+ print(answer)
66
+
67
+ if __name__ == "__main__":
68
+ main()
provenance.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "repository": "mario-rc/emotional-rlaif-dpo-gemma-2-9b-it",
3
+ "base_model": "google/gemma-2-9b-it",
4
+ "base_revision_checked": "11c9b309abf73637e4b6f9a3fa1e92e615547819",
5
+ "base_revision_note": "Revision pinned for this release; historical training revision was not recorded.",
6
+ "method": "DPO",
7
+ "source_run": "dpo_bs64_1ep",
8
+ "source_adapter": "phase3-rlaif-alignment/rlaif-model/rlaif-llama-factory-training/saves/gemma-2-9b-it/lora/dpo_bs64_1ep",
9
+ "adapter_sha256": "6ba739a50256d7a882c1f75cd66075a4e1da9d6f40445edbb5c6b63af2e973c8",
10
+ "seed": 42,
11
+ "sft_initialization": "phase2-sft-alignment/sft-model/sft-llama-factory-training/saves/gemma-2-9b-it/lora/sft_3ep",
12
+ "sft_adapter_sha256": "104d3f4e5e031d27c2b9e906a3b83f079e663d4c7d4ef4aafb6f917b992859ef",
13
+ "lineage": "base -> matching SFT -> DPO",
14
+ "training_config_sha256": "3a03d62cc5302b1c78616237fdf95dd9c879a9e5364bd72fb6523cfd4af82fbd",
15
+ "release_date": "2026-09-16",
16
+ "numerical_evaluation_published": false,
17
+ "tokenizer_release_note": "Vocabulary unchanged; chat_template regenerated from the exact LLaMA-Factory family template so system prompts and generation match training."
18
+ }
requirements.txt ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ torch==2.5.1
2
+ transformers==4.45.2
3
+ peft==0.11.1
4
+ accelerate==0.34.0
5
+ sentencepiece
tokenizer_config.json CHANGED
@@ -2000,10 +2000,10 @@
2000
  "<end_of_turn>"
2001
  ],
2002
  "bos_token": "<bos>",
2003
- "chat_template": "{{ bos_token }}{% if messages[0]['role'] == 'system' %}{{ raise_exception('System role not supported') }}{% endif %}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% if (message['role'] == 'assistant') %}{% set role = 'model' %}{% else %}{% set role = message['role'] %}{% endif %}{{ '<start_of_turn>' + role + '\n' + message['content'] | trim + '<end_of_turn>\n' }}{% endfor %}{% if add_generation_prompt %}{{'<start_of_turn>model\n'}}{% endif %}",
2004
  "clean_up_tokenization_spaces": false,
2005
  "eos_token": "<eos>",
2006
- "model_max_length": 1000000000000000019884624838656,
2007
  "pad_token": "<pad>",
2008
  "padding_side": "right",
2009
  "sp_model_kwargs": {},
 
2000
  "<end_of_turn>"
2001
  ],
2002
  "bos_token": "<bos>",
2003
+ "chat_template": "{{ '<bos>' }}{% if messages[0]['role'] == 'system' %}{% set loop_messages = messages[1:] %}{% set system_message = messages[0]['content'] %}{% else %}{% set loop_messages = messages %}{% endif %}{% if system_message is defined %}{{ system_message }}{% endif %}{% for message in loop_messages %}{% set content = message['content'] %}{% if message['role'] == 'user' %}{{ '<start_of_turn>user\n' + content + '<end_of_turn>\n<start_of_turn>model\n' }}{% elif message['role'] == 'assistant' %}{{ content + '<end_of_turn>\n' }}{% endif %}{% endfor %}",
2004
  "clean_up_tokenization_spaces": false,
2005
  "eos_token": "<eos>",
2006
+ "model_max_length": 2048,
2007
  "pad_token": "<pad>",
2008
  "padding_side": "right",
2009
  "sp_model_kwargs": {},
train_results.json DELETED
@@ -1,8 +0,0 @@
1
- {
2
- "epoch": 0.99996003996004,
3
- "total_flos": 6.814329228177064e+18,
4
- "train_loss": 0.41961536931869625,
5
- "train_runtime": 114854.9374,
6
- "train_samples_per_second": 0.872,
7
- "train_steps_per_second": 0.109
8
- }
 
 
 
 
 
 
 
 
 
trainer_log.jsonl DELETED
The diff for this file is too large to render. See raw diff
 
trainer_state.json DELETED
The diff for this file is too large to render. See raw diff
 
training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:c06fe1f1f49ba1d8abd39b6e211d06afdaec414f07cd625cdd9521622cb8f576
3
- size 5368
 
 
 
 
training_config.json ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "stage": "dpo",
3
+ "template": "gemma",
4
+ "cutoff_len": 2048,
5
+ "dataset": "dpo_preference_dataset",
6
+ "per_device_train_batch_size": 1,
7
+ "gradient_accumulation_steps": 64,
8
+ "learning_rate": 5e-06,
9
+ "num_train_epochs": 1.0,
10
+ "lr_scheduler_type": "cosine",
11
+ "warmup_ratio": 0.1,
12
+ "bf16": true,
13
+ "pref_beta": 0.1,
14
+ "pref_loss": "sigmoid",
15
+ "weight_decay": 0.0,
16
+ "max_grad_norm": 1.0,
17
+ "seed": 42,
18
+ "data_seed": null,
19
+ "max_steps": -1,
20
+ "optim": "adamw_torch",
21
+ "lora_rank": 8,
22
+ "lora_alpha": 16,
23
+ "lora_dropout": 0.0,
24
+ "target_modules": [
25
+ "down_proj",
26
+ "q_proj",
27
+ "v_proj",
28
+ "gate_proj",
29
+ "o_proj",
30
+ "k_proj",
31
+ "up_proj"
32
+ ],
33
+ "sft_initialization": "../../../phase2-sft-alignment/sft-model/sft-llama-factory-training/saves/gemma-2-9b-it/lora/sft_3ep"
34
+ }
training_eval_loss.png DELETED
Binary file (25.9 kB)
 
training_loss.png DELETED
Binary file (50 kB)
 
training_rewards_accuracies.png DELETED
Binary file (52.3 kB)