mixtral-8x7b-moe-nq-finetuned / training_state.json
vvijayk's picture
Upload fine-tuned MPT-7B-MoE model
6ff2e53 verified
Raw History Blame Contribute Delete
739 Bytes
{
"global_step": 152,
"epoch": 2,
"completed_epoch": 2,
"args": {
"model_id": "mistralai/Mixtral-8x7B-v0.1",
"model_revision": null,
"data_file": "nq_annotated_moe_balanced.jsonl",
"output_dir": "./mixtral_moe_supervised",
"max_seq_length": 64,
"max_target_length": 32,
"learning_rate": 1e-05,
"epochs": 2,
"per_device_batch_size": 1,
"gradient_accumulation_steps": 8,
"warmup_steps": 200,
"max_grad_norm": 1.0,
"seed": 42,
"logging_steps": 10,
"save_steps": 500,
"max_samples": null,
"convert_to_moe": false,
"resume_from_checkpoint": null,
"routing_loss_weight": 0.1,
"disable_supervised_routing": false,
"label_embedding_temperature": 1.0
}
}