Jeesup commited on
Commit
99efb1b
·
verified ·
1 Parent(s): 9254674

LoRA adapter + metrics (bit-width study)

Browse files
README.md ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: peft
3
+ base_model: meta-llama/Llama-3.2-1B
4
+ tags:
5
+ - lora
6
+ - peft
7
+ - quantization
8
+ - glue
9
+ - sst2
10
+ ---
11
+
12
+ # llama32-1B-sst2-int8-lora-seed42
13
+
14
+ LoRA adapter trained on GLUE **SST2** on top of a
15
+ **int8** backbone of `meta-llama/Llama-3.2-1B`.
16
+
17
+ Part of a controlled study of whether the backbone bit-width changes what a LoRA
18
+ adapter learns. For a given (model size, seed) the adapter initialisation is
19
+ **identical** across the bf16 / int8 / nf4 arms, and the data order, optimiser,
20
+ schedule and LoRA hyperparameters are held fixed — so any difference in the
21
+ learned update is attributable to the backbone.
22
+
23
+ ## Result
24
+
25
+ | metric | validation | test |
26
+ |---|---|---|
27
+ | accuracy | 0.9640 | 0.9541 |
28
+ | macro-F1 | 0.9635 | 0.9541 |
29
+ | loss | 0.1229 | 0.1597 |
30
+
31
+ Test-set majority-class baseline: 0.5092
32
+
33
+ - peak GPU memory: 2.97 GiB
34
+ - training time: 47.9 min (936 steps)
35
+ - GPU: NVIDIA GeForce RTX 4090
36
+
37
+ ## Setup
38
+
39
+ - seed: `42` · adapter init: `shared:lora_init_1B_seed42.pt:128tensors`
40
+ - LoRA: r=16, alpha=32, dropout=0.0, bias=none,
41
+ target_modules=['q_proj', 'k_proj', 'v_proj', 'o_proj']
42
+ - trainable params: 3,407,872
43
+ - epochs 3, lr 0.0002,
44
+ max_len 256, batch 4
45
+ x grad_accum 16,
46
+ cosine schedule, warmup 0.03
47
+
48
+ ## Prompt format
49
+
50
+ Trained as causal LM with the loss on the answer letter only (prompt tokens
51
+ masked to -100):
52
+
53
+ ```
54
+ Sentence: ...
55
+
56
+ Is the sentiment of this sentence positive or negative?
57
+
58
+ A. Negative
59
+ B. Positive
60
+
61
+ Answer:
62
+ ```
63
+
64
+ Evaluated by conditional likelihood over the answer letters
65
+ (Negative, Positive).
66
+
67
+ > GLUE `test` is unlabeled, so the official `validation` split is used as TEST
68
+ > and the validation set is carved from `train` (disjoint).
adapter_config.json ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "meta-llama/Llama-3.2-1B",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.0,
22
+ "lora_ga_config": null,
23
+ "megatron_config": null,
24
+ "megatron_core": "megatron.core",
25
+ "modules_to_save": null,
26
+ "peft_type": "LORA",
27
+ "peft_version": "0.19.1",
28
+ "qalora_group_size": 16,
29
+ "r": 16,
30
+ "rank_pattern": {},
31
+ "revision": null,
32
+ "target_modules": [
33
+ "k_proj",
34
+ "q_proj",
35
+ "v_proj",
36
+ "o_proj"
37
+ ],
38
+ "target_parameters": null,
39
+ "task_type": "CAUSAL_LM",
40
+ "trainable_token_indices": null,
41
+ "use_bdlora": null,
42
+ "use_dora": false,
43
+ "use_qalora": false,
44
+ "use_rslora": false
45
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b55c5d1dbb763bd4d20cb63f8e884c984ef6628488a7cb51c984a249cfadb37d
3
+ size 13648488
run_metrics.json ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "run_name": "llama32-1B-sst2-int8-lora-seed42",
3
+ "model_size": "1B",
4
+ "model_id": "meta-llama/Llama-3.2-1B",
5
+ "task": "sst2",
6
+ "glue_config": "sst2",
7
+ "n_classes": 2,
8
+ "bitwidth": "int8",
9
+ "seed": 42,
10
+ "lora_init_source": "shared:lora_init_1B_seed42.pt:128tensors",
11
+ "trainable_params": 3407872,
12
+ "total_params": 1239222272,
13
+ "train": {
14
+ "steps": 936,
15
+ "epochs": 2.992,
16
+ "train_runtime_sec": 2875.6240453720093,
17
+ "train_loss": 0.12485106690571858,
18
+ "n_train_examples": 20000,
19
+ "n_train_tokens_per_epoch": 720480,
20
+ "throughput_samples_per_sec": 20.809396171347814,
21
+ "throughput_tokens_per_sec": 749.6376876766336
22
+ },
23
+ "peak_gpu_mem_gib": {
24
+ "allocated": 2.9748597145080566,
25
+ "reserved": 6.1640625
26
+ },
27
+ "validation": {
28
+ "accuracy": 0.964,
29
+ "macro_f1": 0.963456080299173,
30
+ "loss": 0.1228773279953748,
31
+ "n": 1000,
32
+ "n_classes": 2,
33
+ "majority_baseline": 0.56,
34
+ "pred_dist": {
35
+ "A": 438,
36
+ "B": 562
37
+ },
38
+ "true_dist": {
39
+ "A": 440,
40
+ "B": 560
41
+ }
42
+ },
43
+ "test": {
44
+ "accuracy": 0.9541284403669725,
45
+ "macro_f1": 0.954104296932567,
46
+ "loss": 0.15973422233305803,
47
+ "n": 872,
48
+ "n_classes": 2,
49
+ "majority_baseline": 0.5091743119266054,
50
+ "pred_dist": {
51
+ "A": 424,
52
+ "B": 448
53
+ },
54
+ "true_dist": {
55
+ "A": 428,
56
+ "B": 444
57
+ }
58
+ },
59
+ "hyperparams": {
60
+ "num_train_epochs": 3,
61
+ "learning_rate": 0.0002,
62
+ "max_length": 256,
63
+ "per_device_train_batch_size": 4,
64
+ "per_device_eval_batch_size": 8,
65
+ "gradient_accumulation_steps": 16,
66
+ "warmup_ratio": 0.03,
67
+ "lr_scheduler_type": "cosine",
68
+ "gradient_checkpointing": true
69
+ },
70
+ "lora": {
71
+ "r": 16,
72
+ "alpha": 32,
73
+ "dropout": 0.0,
74
+ "bias": "none",
75
+ "target_modules": [
76
+ "q_proj",
77
+ "k_proj",
78
+ "v_proj",
79
+ "o_proj"
80
+ ]
81
+ },
82
+ "env": {
83
+ "torch": "2.4.1+cu121",
84
+ "cuda": "12.1",
85
+ "gpu": "NVIDIA GeForce RTX 4090",
86
+ "slurm_job": "1930325",
87
+ "node": "node40"
88
+ },
89
+ "debug_caps": {
90
+ "max_train": null,
91
+ "max_eval": null,
92
+ "max_steps": -1
93
+ }
94
+ }
train_config.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_id": "meta-llama/Llama-3.2-1B",
3
+ "task": "sst2",
4
+ "glue_config": "sst2",
5
+ "bitwidth": "int8",
6
+ "seed": 42,
7
+ "lora": {
8
+ "r": 16,
9
+ "alpha": 32,
10
+ "dropout": 0.0,
11
+ "bias": "none",
12
+ "target_modules": [
13
+ "q_proj",
14
+ "k_proj",
15
+ "v_proj",
16
+ "o_proj"
17
+ ]
18
+ },
19
+ "hyperparams": {
20
+ "num_train_epochs": 3,
21
+ "learning_rate": 0.0002,
22
+ "max_length": 256,
23
+ "per_device_train_batch_size": 4,
24
+ "per_device_eval_batch_size": 8,
25
+ "gradient_accumulation_steps": 16,
26
+ "warmup_ratio": 0.03,
27
+ "lr_scheduler_type": "cosine",
28
+ "gradient_checkpointing": true
29
+ },
30
+ "lora_init_source": "shared:lora_init_1B_seed42.pt:128tensors"
31
+ }