Instructions to use Jeesup/llama32-3B-rte-nf4-lora-seed43 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use Jeesup/llama32-3B-rte-nf4-lora-seed43 with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-3.2-3B") model = PeftModel.from_pretrained(base_model, "Jeesup/llama32-3B-rte-nf4-lora-seed43") - Notebooks
- Google Colab
- Kaggle
File size: 2,119 Bytes
38b095b | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 | {
"run_name": "llama32-3B-rte-nf4-lora-seed43",
"model_size": "3B",
"model_id": "meta-llama/Llama-3.2-3B",
"task": "rte",
"glue_config": "rte",
"n_classes": 2,
"bitwidth": "nf4",
"seed": 43,
"lora_init_source": "shared:lora_init_3B_seed43.pt:224tensors",
"trainable_params": 9175040,
"total_params": 1812638720,
"train": {
"steps": 105,
"epochs": 2.9411764705882355,
"train_runtime_sec": 385.316321849823,
"train_loss": 0.476223771912711,
"n_train_examples": 2241,
"n_train_tokens_per_epoch": 219447,
"throughput_samples_per_sec": 17.105884430084295,
"throughput_tokens_per_sec": 1675.071405858415
},
"peak_gpu_mem_gib": {
"allocated": 5.5141496658325195,
"reserved": 9.181640625
},
"validation": {
"accuracy": 0.8433734939759037,
"macro_f1": 0.8428767899037295,
"loss": 0.36798715854744357,
"n": 249,
"n_classes": 2,
"majority_baseline": 0.5140562248995983,
"pred_dist": {
"A": 142,
"B": 107
},
"true_dist": {
"A": 121,
"B": 128
}
},
"test": {
"accuracy": 0.8483754512635379,
"macro_f1": 0.8471302428256071,
"loss": 0.44474018362454987,
"n": 277,
"n_classes": 2,
"majority_baseline": 0.5270758122743683,
"pred_dist": {
"A": 156,
"B": 121
},
"true_dist": {
"A": 146,
"B": 131
}
},
"hyperparams": {
"num_train_epochs": 3,
"learning_rate": 0.0002,
"max_length": 256,
"per_device_train_batch_size": 4,
"per_device_eval_batch_size": 8,
"gradient_accumulation_steps": 16,
"warmup_ratio": 0.03,
"lr_scheduler_type": "cosine",
"gradient_checkpointing": true
},
"lora": {
"r": 16,
"alpha": 32,
"dropout": 0.0,
"bias": "none",
"target_modules": [
"q_proj",
"k_proj",
"v_proj",
"o_proj"
]
},
"env": {
"torch": "2.4.1+cu121",
"cuda": "12.1",
"gpu": "NVIDIA GeForce RTX 4090",
"slurm_job": "1932942",
"node": "node39"
},
"debug_caps": {
"max_train": null,
"max_eval": null,
"max_steps": -1
}
} |