Instructions to use Jeesup/llama32-3B-rte-nf4-lora-seed43 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use Jeesup/llama32-3B-rte-nf4-lora-seed43 with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-3.2-3B") model = PeftModel.from_pretrained(base_model, "Jeesup/llama32-3B-rte-nf4-lora-seed43") - Notebooks
- Google Colab
- Kaggle
Download run_metrics.json from Jeesup/llama32-3B-rte-nf4-lora-seed43: direct link, hf CLI and curl.
- Browser
- Download file 2.12 kB
-
https://huggingface.co/Jeesup/llama32-3B-rte-nf4-lora-seed43/resolve/main/run_metrics.json
- Command line
-
hf download hf://Jeesup/llama32-3B-rte-nf4-lora-seed43/run_metrics.json
-
curl -L -o run_metrics.json https://huggingface.co/Jeesup/llama32-3B-rte-nf4-lora-seed43/resolve/main/run_metrics.json
2.12 kB
| { | |
| "run_name": "llama32-3B-rte-nf4-lora-seed43", | |
| "model_size": "3B", | |
| "model_id": "meta-llama/Llama-3.2-3B", | |
| "task": "rte", | |
| "glue_config": "rte", | |
| "n_classes": 2, | |
| "bitwidth": "nf4", | |
| "seed": 43, | |
| "lora_init_source": "shared:lora_init_3B_seed43.pt:224tensors", | |
| "trainable_params": 9175040, | |
| "total_params": 1812638720, | |
| "train": { | |
| "steps": 105, | |
| "epochs": 2.9411764705882355, | |
| "train_runtime_sec": 385.316321849823, | |
| "train_loss": 0.476223771912711, | |
| "n_train_examples": 2241, | |
| "n_train_tokens_per_epoch": 219447, | |
| "throughput_samples_per_sec": 17.105884430084295, | |
| "throughput_tokens_per_sec": 1675.071405858415 | |
| }, | |
| "peak_gpu_mem_gib": { | |
| "allocated": 5.5141496658325195, | |
| "reserved": 9.181640625 | |
| }, | |
| "validation": { | |
| "accuracy": 0.8433734939759037, | |
| "macro_f1": 0.8428767899037295, | |
| "loss": 0.36798715854744357, | |
| "n": 249, | |
| "n_classes": 2, | |
| "majority_baseline": 0.5140562248995983, | |
| "pred_dist": { | |
| "A": 142, | |
| "B": 107 | |
| }, | |
| "true_dist": { | |
| "A": 121, | |
| "B": 128 | |
| } | |
| }, | |
| "test": { | |
| "accuracy": 0.8483754512635379, | |
| "macro_f1": 0.8471302428256071, | |
| "loss": 0.44474018362454987, | |
| "n": 277, | |
| "n_classes": 2, | |
| "majority_baseline": 0.5270758122743683, | |
| "pred_dist": { | |
| "A": 156, | |
| "B": 121 | |
| }, | |
| "true_dist": { | |
| "A": 146, | |
| "B": 131 | |
| } | |
| }, | |
| "hyperparams": { | |
| "num_train_epochs": 3, | |
| "learning_rate": 0.0002, | |
| "max_length": 256, | |
| "per_device_train_batch_size": 4, | |
| "per_device_eval_batch_size": 8, | |
| "gradient_accumulation_steps": 16, | |
| "warmup_ratio": 0.03, | |
| "lr_scheduler_type": "cosine", | |
| "gradient_checkpointing": true | |
| }, | |
| "lora": { | |
| "r": 16, | |
| "alpha": 32, | |
| "dropout": 0.0, | |
| "bias": "none", | |
| "target_modules": [ | |
| "q_proj", | |
| "k_proj", | |
| "v_proj", | |
| "o_proj" | |
| ] | |
| }, | |
| "env": { | |
| "torch": "2.4.1+cu121", | |
| "cuda": "12.1", | |
| "gpu": "NVIDIA GeForce RTX 4090", | |
| "slurm_job": "1932942", | |
| "node": "node39" | |
| }, | |
| "debug_caps": { | |
| "max_train": null, | |
| "max_eval": null, | |
| "max_steps": -1 | |
| } | |
| } |