Jeesup commited on
Commit
fbe382f
·
verified ·
1 Parent(s): 43b1fc3

LoRA adapter + metrics (bit-width study)

Browse files
README.md ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: peft
3
+ base_model: meta-llama/Llama-3.2-3B
4
+ tags:
5
+ - lora
6
+ - peft
7
+ - quantization
8
+ - glue
9
+ - rte
10
+ ---
11
+
12
+ # llama32-3B-rte-int8-lora-seed44
13
+
14
+ LoRA adapter trained on GLUE **RTE** on top of a
15
+ **int8** backbone of `meta-llama/Llama-3.2-3B`.
16
+
17
+ Part of a controlled study of whether the backbone bit-width changes what a LoRA
18
+ adapter learns. For a given (model size, seed) the adapter initialisation is
19
+ **identical** across the bf16 / int8 / nf4 arms, and the data order, optimiser,
20
+ schedule and LoRA hyperparameters are held fixed — so any difference in the
21
+ learned update is attributable to the backbone.
22
+
23
+ ## Result
24
+
25
+ | metric | validation | test |
26
+ |---|---|---|
27
+ | accuracy | 0.8876 | 0.8556 |
28
+ | macro-F1 | 0.8875 | 0.8546 |
29
+ | loss | 0.2930 | 0.3882 |
30
+
31
+ Test-set majority-class baseline: 0.5271
32
+
33
+ - peak GPU memory: 6.81 GiB
34
+ - training time: 11.3 min (105 steps)
35
+ - GPU: NVIDIA GeForce RTX 4090
36
+
37
+ ## Setup
38
+
39
+ - seed: `44` · adapter init: `shared:lora_init_3B_seed44.pt:224tensors`
40
+ - LoRA: r=16, alpha=32, dropout=0.0, bias=none,
41
+ target_modules=['q_proj', 'k_proj', 'v_proj', 'o_proj']
42
+ - trainable params: 9,175,040
43
+ - epochs 3, lr 0.0002,
44
+ max_len 256, batch 4
45
+ x grad_accum 16,
46
+ cosine schedule, warmup 0.03
47
+
48
+ ## Prompt format
49
+
50
+ Trained as causal LM with the loss on the answer letter only (prompt tokens
51
+ masked to -100):
52
+
53
+ ```
54
+ Premise: ...
55
+ Hypothesis: ...
56
+
57
+ Does the premise entail the hypothesis?
58
+
59
+ A. Entailment
60
+ B. Not entailment
61
+
62
+ Answer:
63
+ ```
64
+
65
+ Evaluated by conditional likelihood over the answer letters
66
+ (Entailment, Not entailment).
67
+
68
+ > GLUE `test` is unlabeled, so the official `validation` split is used as TEST
69
+ > and the validation set is carved from `train` (disjoint).
adapter_config.json ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "meta-llama/Llama-3.2-3B",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0,
22
+ "lora_ga_config": null,
23
+ "megatron_config": null,
24
+ "megatron_core": "megatron.core",
25
+ "modules_to_save": null,
26
+ "peft_type": "LORA",
27
+ "peft_version": "0.19.1",
28
+ "qalora_group_size": 16,
29
+ "r": 16,
30
+ "rank_pattern": {},
31
+ "revision": null,
32
+ "target_modules": [
33
+ "v_proj",
34
+ "q_proj",
35
+ "k_proj",
36
+ "o_proj"
37
+ ],
38
+ "target_parameters": null,
39
+ "task_type": "CAUSAL_LM",
40
+ "trainable_token_indices": null,
41
+ "use_bdlora": null,
42
+ "use_dora": false,
43
+ "use_qalora": false,
44
+ "use_rslora": false
45
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ee3e38255985a3e276f30ffdf16cca34abe7a0f296e37533196722fc5e254ca
3
+ size 36730224
run_metrics.json ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "run_name": "llama32-3B-rte-int8-lora-seed44",
3
+ "model_size": "3B",
4
+ "model_id": "meta-llama/Llama-3.2-3B",
5
+ "task": "rte",
6
+ "glue_config": "rte",
7
+ "n_classes": 2,
8
+ "bitwidth": "int8",
9
+ "seed": 44,
10
+ "lora_init_source": "shared:lora_init_3B_seed44.pt:224tensors",
11
+ "trainable_params": 9175040,
12
+ "total_params": 3221924864,
13
+ "train": {
14
+ "steps": 105,
15
+ "epochs": 2.9411764705882355,
16
+ "train_runtime_sec": 678.9310655593872,
17
+ "train_loss": 0.42380388123648505,
18
+ "n_train_examples": 2241,
19
+ "n_train_tokens_per_epoch": 219447,
20
+ "throughput_samples_per_sec": 9.708167448719719,
21
+ "throughput_tokens_per_sec": 950.6596261129835
22
+ },
23
+ "peak_gpu_mem_gib": {
24
+ "allocated": 6.807776927947998,
25
+ "reserved": 11.232421875
26
+ },
27
+ "validation": {
28
+ "accuracy": 0.8875502008032129,
29
+ "macro_f1": 0.8875483870967742,
30
+ "loss": 0.293006686800455,
31
+ "n": 249,
32
+ "n_classes": 2,
33
+ "majority_baseline": 0.5140562248995983,
34
+ "pred_dist": {
35
+ "A": 129,
36
+ "B": 120
37
+ },
38
+ "true_dist": {
39
+ "A": 121,
40
+ "B": 128
41
+ }
42
+ },
43
+ "test": {
44
+ "accuracy": 0.855595667870036,
45
+ "macro_f1": 0.8545931758530183,
46
+ "loss": 0.38815017738497215,
47
+ "n": 277,
48
+ "n_classes": 2,
49
+ "majority_baseline": 0.5270758122743683,
50
+ "pred_dist": {
51
+ "A": 154,
52
+ "B": 123
53
+ },
54
+ "true_dist": {
55
+ "A": 146,
56
+ "B": 131
57
+ }
58
+ },
59
+ "hyperparams": {
60
+ "num_train_epochs": 3,
61
+ "learning_rate": 0.0002,
62
+ "max_length": 256,
63
+ "per_device_train_batch_size": 4,
64
+ "per_device_eval_batch_size": 8,
65
+ "gradient_accumulation_steps": 16,
66
+ "warmup_ratio": 0.03,
67
+ "lr_scheduler_type": "cosine",
68
+ "gradient_checkpointing": true
69
+ },
70
+ "lora": {
71
+ "r": 16,
72
+ "alpha": 32,
73
+ "dropout": 0.0,
74
+ "bias": "none",
75
+ "target_modules": [
76
+ "q_proj",
77
+ "k_proj",
78
+ "v_proj",
79
+ "o_proj"
80
+ ]
81
+ },
82
+ "env": {
83
+ "torch": "2.4.1+cu121",
84
+ "cuda": "12.1",
85
+ "gpu": "NVIDIA GeForce RTX 4090",
86
+ "slurm_job": "1932576",
87
+ "node": "node32"
88
+ },
89
+ "debug_caps": {
90
+ "max_train": null,
91
+ "max_eval": null,
92
+ "max_steps": -1
93
+ }
94
+ }
train_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_id": "meta-llama/Llama-3.2-3B",
3
+ "task": "rte",
4
+ "glue_config": "rte",
5
+ "bitwidth": "int8",
6
+ "seed": 44,
7
+ "lora": {
8
+ "r": 16,
9
+ "alpha": 32,
10
+ "dropout": 0.0,
11
+ "bias": "none",
12
+ "target_modules": [
13
+ "q_proj",
14
+ "k_proj",
15
+ "v_proj",
16
+ "o_proj"
17
+ ]
18
+ },
19
+ "hyperparams": {
20
+ "num_train_epochs": 3,
21
+ "learning_rate": 0.0002,
22
+ "max_length": 256,
23
+ "per_device_train_batch_size": 4,
24
+ "per_device_eval_batch_size": 8,
25
+ "gradient_accumulation_steps": 16,
26
+ "warmup_ratio": 0.03,
27
+ "lr_scheduler_type": "cosine",
28
+ "gradient_checkpointing": true
29
+ },
30
+ "lora_init_source": "shared:lora_init_3B_seed44.pt:224tensors"
31
+ }