{ "model_id": "meta-llama/Llama-3.2-3B", "task": "rte", "glue_config": "rte", "bitwidth": "nf4", "seed": 43, "lora": { "r": 16, "alpha": 32, "dropout": 0.0, "bias": "none", "target_modules": [ "q_proj", "k_proj", "v_proj", "o_proj" ] }, "hyperparams": { "num_train_epochs": 3, "learning_rate": 0.0002, "max_length": 256, "per_device_train_batch_size": 4, "per_device_eval_batch_size": 8, "gradient_accumulation_steps": 16, "warmup_ratio": 0.03, "lr_scheduler_type": "cosine", "gradient_checkpointing": true }, "lora_init_source": "shared:lora_init_3B_seed43.pt:224tensors" }