Download training_state.json from vvijayk/mixtral-8x7b-moe-nq-finetuned: direct link, hf CLI and curl.
- Browser
- Download file 739 Bytes
-
https://huggingface.co/vvijayk/mixtral-8x7b-moe-nq-finetuned/resolve/main/training_state.json
- Command line
-
hf download hf://vvijayk/mixtral-8x7b-moe-nq-finetuned/training_state.json
-
curl -L -o training_state.json https://huggingface.co/vvijayk/mixtral-8x7b-moe-nq-finetuned/resolve/main/training_state.json
739 Bytes
| { | |
| "global_step": 152, | |
| "epoch": 2, | |
| "completed_epoch": 2, | |
| "args": { | |
| "model_id": "mistralai/Mixtral-8x7B-v0.1", | |
| "model_revision": null, | |
| "data_file": "nq_annotated_moe_balanced.jsonl", | |
| "output_dir": "./mixtral_moe_supervised", | |
| "max_seq_length": 64, | |
| "max_target_length": 32, | |
| "learning_rate": 1e-05, | |
| "epochs": 2, | |
| "per_device_batch_size": 1, | |
| "gradient_accumulation_steps": 8, | |
| "warmup_steps": 200, | |
| "max_grad_norm": 1.0, | |
| "seed": 42, | |
| "logging_steps": 10, | |
| "save_steps": 500, | |
| "max_samples": null, | |
| "convert_to_moe": false, | |
| "resume_from_checkpoint": null, | |
| "routing_loss_weight": 0.1, | |
| "disable_supervised_routing": false, | |
| "label_embedding_temperature": 1.0 | |
| } | |
| } |