Instructions to use Jeesup/llama32-3B-rte-bf16-lora-seed44 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use Jeesup/llama32-3B-rte-bf16-lora-seed44 with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("meta-llama/Llama-3.2-3B") model = PeftModel.from_pretrained(base_model, "Jeesup/llama32-3B-rte-bf16-lora-seed44") - Notebooks
- Google Colab
- Kaggle
Download run_metrics.json from Jeesup/llama32-3B-rte-bf16-lora-seed44: direct link, hf CLI and curl.
- Browser
- Download file 2.12 kB
-
https://huggingface.co/Jeesup/llama32-3B-rte-bf16-lora-seed44/resolve/main/run_metrics.json
- Command line
-
hf download hf://Jeesup/llama32-3B-rte-bf16-lora-seed44/run_metrics.json
-
curl -L -o run_metrics.json https://huggingface.co/Jeesup/llama32-3B-rte-bf16-lora-seed44/resolve/main/run_metrics.json
2.12 kB
| { | |
| "run_name": "llama32-3B-rte-bf16-lora-seed44", | |
| "model_size": "3B", | |
| "model_id": "meta-llama/Llama-3.2-3B", | |
| "task": "rte", | |
| "glue_config": "rte", | |
| "n_classes": 2, | |
| "bitwidth": "bf16", | |
| "seed": 44, | |
| "lora_init_source": "shared:lora_init_3B_seed44.pt:224tensors", | |
| "trainable_params": 9175040, | |
| "total_params": 3221924864, | |
| "train": { | |
| "steps": 105, | |
| "epochs": 2.9411764705882355, | |
| "train_runtime_sec": 225.0290515422821, | |
| "train_loss": 0.4039383820125035, | |
| "n_train_examples": 2241, | |
| "n_train_tokens_per_epoch": 219447, | |
| "throughput_samples_per_sec": 29.290335738493653, | |
| "throughput_tokens_per_sec": 2868.217896834099 | |
| }, | |
| "peak_gpu_mem_gib": { | |
| "allocated": 7.778663635253906, | |
| "reserved": 14.548828125 | |
| }, | |
| "validation": { | |
| "accuracy": 0.8594377510040161, | |
| "macro_f1": 0.8594014680971204, | |
| "loss": 0.3213284973159851, | |
| "n": 249, | |
| "n_classes": 2, | |
| "majority_baseline": 0.5140562248995983, | |
| "pred_dist": { | |
| "A": 132, | |
| "B": 117 | |
| }, | |
| "true_dist": { | |
| "A": 121, | |
| "B": 128 | |
| } | |
| }, | |
| "test": { | |
| "accuracy": 0.8447653429602888, | |
| "macro_f1": 0.8442456814823533, | |
| "loss": 0.4106067032805419, | |
| "n": 277, | |
| "n_classes": 2, | |
| "majority_baseline": 0.5270758122743683, | |
| "pred_dist": { | |
| "A": 147, | |
| "B": 130 | |
| }, | |
| "true_dist": { | |
| "A": 146, | |
| "B": 131 | |
| } | |
| }, | |
| "hyperparams": { | |
| "num_train_epochs": 3, | |
| "learning_rate": 0.0002, | |
| "max_length": 256, | |
| "per_device_train_batch_size": 4, | |
| "per_device_eval_batch_size": 8, | |
| "gradient_accumulation_steps": 16, | |
| "warmup_ratio": 0.03, | |
| "lr_scheduler_type": "cosine", | |
| "gradient_checkpointing": true | |
| }, | |
| "lora": { | |
| "r": 16, | |
| "alpha": 32, | |
| "dropout": 0.0, | |
| "bias": "none", | |
| "target_modules": [ | |
| "q_proj", | |
| "k_proj", | |
| "v_proj", | |
| "o_proj" | |
| ] | |
| }, | |
| "env": { | |
| "torch": "2.4.1+cu121", | |
| "cuda": "12.1", | |
| "gpu": "NVIDIA GeForce RTX 4090", | |
| "slurm_job": "1932474", | |
| "node": "node39" | |
| }, | |
| "debug_caps": { | |
| "max_train": null, | |
| "max_eval": null, | |
| "max_steps": -1 | |
| } | |
| } |