{ "repository": "mario-rc/emotional-rlaif-dpo-gemma-2-9b-it", "base_model": "google/gemma-2-9b-it", "base_revision_checked": "11c9b309abf73637e4b6f9a3fa1e92e615547819", "base_revision_note": "Revision pinned for this release; historical training revision was not recorded.", "method": "DPO", "source_run": "dpo_bs64_1ep", "source_adapter": "phase3-rlaif-alignment/rlaif-model/rlaif-llama-factory-training/saves/gemma-2-9b-it/lora/dpo_bs64_1ep", "adapter_sha256": "6ba739a50256d7a882c1f75cd66075a4e1da9d6f40445edbb5c6b63af2e973c8", "seed": 42, "sft_initialization": "phase2-sft-alignment/sft-model/sft-llama-factory-training/saves/gemma-2-9b-it/lora/sft_3ep", "sft_adapter_sha256": "104d3f4e5e031d27c2b9e906a3b83f079e663d4c7d4ef4aafb6f917b992859ef", "lineage": "base -> matching SFT -> DPO", "training_config_sha256": "3a03d62cc5302b1c78616237fdf95dd9c879a9e5364bd72fb6523cfd4af82fbd", "release_date": "2026-09-16", "numerical_evaluation_published": false, "tokenizer_release_note": "Vocabulary unchanged; chat_template regenerated from the exact LLaMA-Factory family template so system prompts and generation match training." }