nbtpj commited on
Commit
6c9c052
·
verified ·
1 Parent(s): c362895

Upload best model checkpoint

Browse files
config.json ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "activation_function": "gelu_new",
3
+ "architectures": [
4
+ "GPT2LMHeadModel"
5
+ ],
6
+ "attn_pdrop": 0.1,
7
+ "bos_token_id": 50256,
8
+ "dtype": "float32",
9
+ "embd_pdrop": 0.1,
10
+ "eos_token_id": 50256,
11
+ "initializer_range": 0.02,
12
+ "layer_norm_epsilon": 1e-05,
13
+ "model_type": "gpt2",
14
+ "n_ctx": 1024,
15
+ "n_embd": 768,
16
+ "n_head": 12,
17
+ "n_inner": null,
18
+ "n_layer": 12,
19
+ "n_positions": 1024,
20
+ "reorder_and_upcast_attn": false,
21
+ "resid_pdrop": 0.1,
22
+ "scale_attn_by_inverse_layer_idx": false,
23
+ "scale_attn_weights": true,
24
+ "summary_activation": null,
25
+ "summary_first_dropout": 0.1,
26
+ "summary_proj_to_labels": true,
27
+ "summary_type": "cls_index",
28
+ "summary_use_proj": true,
29
+ "task_specific_params": {
30
+ "text-generation": {
31
+ "do_sample": true,
32
+ "max_length": 50
33
+ }
34
+ },
35
+ "transformers_version": "4.57.3",
36
+ "use_cache": true,
37
+ "vocab_size": 50257
38
+ }
generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "bos_token_id": 50256,
4
+ "eos_token_id": 50256,
5
+ "transformers_version": "4.57.3"
6
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
metrics.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "rl_info/A2G": -0.6658737659454346,
3
+ "rl_info/entropy": 0.7045262455940247,
4
+ "rl_info/total_token": 650.0,
5
+ "rl_info/advantage_b4_norm": -2206.338134765625,
6
+ "rl_info/advantage_after_gnorm": -1.0411098003387451,
7
+ "rl_info/kl_w_ref": -0.0003157879982609302,
8
+ "rl_info/clip_ratio": 0.936923086643219,
9
+ "train/rl_loss": 66.5872802734375,
10
+ "train/total_loss": 8.323410034179688,
11
+ "xsum/rouge1": 0.17478446053644453,
12
+ "xsum/rouge2": 0.0270244843023609,
13
+ "xsum/rougeL": 0.13926067850589546,
14
+ "xsum/rougeLsum": 0.14284803751147157,
15
+ "xsum/bertscore_precision": 0.7250652085152888,
16
+ "xsum/bertscore_recall": 0.7281463338055265,
17
+ "xsum/bertscore_f1": 0.7262238959629153,
18
+ "eval_agg/avg_all_rougef": 0.12097941521404312,
19
+ "eval_agg/avg_all_bertf": 0.7262238959629153,
20
+ "eval_agg/avg_all": 0.4236016555884792,
21
+ "num_rl_rollout": 250,
22
+ "lm_epoch": 0,
23
+ "rl_epoch": 0,
24
+ "step": 2000,
25
+ "total_data_token": 3685864,
26
+ "total_rl_token": 8807121,
27
+ "total_lm_token": 0,
28
+ "total_token": 8807121,
29
+ "completed_steps": 2000,
30
+ "tune_objective": 0.41038940313189515
31
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5dc913c5fe263bf822ac5cea8b4d84459769d9fc569cda85f78bdccc846ddd7e
3
+ size 497774208
special_tokens_map.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": "<|endoftext|>",
3
+ "eos_token": "<|endoftext|>",
4
+ "pad_token": "<|endoftext|>",
5
+ "unk_token": "<|endoftext|>"
6
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "50256": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": true,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ }
12
+ },
13
+ "bos_token": "<|endoftext|>",
14
+ "clean_up_tokenization_spaces": false,
15
+ "eos_token": "<|endoftext|>",
16
+ "extra_special_tokens": {},
17
+ "model_max_length": 1024,
18
+ "pad_token": "<|endoftext|>",
19
+ "tokenizer_class": "GPT2Tokenizer",
20
+ "unk_token": "<|endoftext|>"
21
+ }
train_configs.json ADDED
@@ -0,0 +1,112 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "dataset_name": "nbtpj/summ_ds_train",
3
+ "dataset_config_name": null,
4
+ "train_split_name": "sim_with_one_golden__xsum_train",
5
+ "text_template": "{text}\nTL;DR: {summary}",
6
+ "label_col": "summary",
7
+ "freeze_role2": true,
8
+ "only_train_role1": true,
9
+ "model_name_or_path": "gpt2",
10
+ "ref_role1_name_or_path": "nbtpj/summ_gpt2_tldr_xsum",
11
+ "ref_role2_name_or_path": "gpt2",
12
+ "pretrained_role2_name_or_path": "gpt2",
13
+ "config_name": null,
14
+ "vectorizer_path": "/common/home/users/m/mq.nguyen.2023/testcode/SAC_LM/module9_clmv3/vectorizer/wikitext103_tfidf_full.joblib",
15
+ "tokenizer_name": null,
16
+ "use_slow_tokenizer": false,
17
+ "per_device_train_batch_size": 16,
18
+ "per_device_query_rollout_batch_size": 32,
19
+ "per_device_eval_batch_size": 16,
20
+ "vllm_vram_ratio": 0.3,
21
+ "learning_rate": 3e-07,
22
+ "grad_norm": 0.5,
23
+ "weight_decay": 1e-05,
24
+ "max_train_steps": 100000,
25
+ "max_train_rollouts": 100000,
26
+ "gradient_accumulation_steps": 1,
27
+ "lr_scheduler_type": "constant",
28
+ "num_warmup_steps": 200,
29
+ "seed": 0,
30
+ "model_type": null,
31
+ "block_size": 1024,
32
+ "mini_epoch": 2,
33
+ "rollout_game": "baseline3v2",
34
+ "rl_algo": "on_policy",
35
+ "constraint_type": "kl",
36
+ "clamp_update": false,
37
+ "clamp_ref": true,
38
+ "no_is_correction": false,
39
+ "rl_w": 1.0,
40
+ "lm_w": 0.0,
41
+ "n_generate": 4,
42
+ "n_augment": 0,
43
+ "gradient_checkpoint": false,
44
+ "group_relative_norm": true,
45
+ "sample_config": {
46
+ "do_sample": true,
47
+ "min_new_tokens": 5,
48
+ "temperature": 1.0
49
+ },
50
+ "inference_config": {
51
+ "do_sample": true,
52
+ "temperature": 0.0,
53
+ "min_new_tokens": 15,
54
+ "max_new_tokens": 40
55
+ },
56
+ "rollout_config": {
57
+ "accuracy_w": 1000.0,
58
+ "len_pen": 1.0,
59
+ "accuracy_w2": 0.01,
60
+ "len_pen2": 1.0,
61
+ "threshold": 0.002,
62
+ "similarity_fn": "rouge",
63
+ "acc_scale": "log"
64
+ },
65
+ "ent_coef": 0.0001,
66
+ "beta_coef": "0.1",
67
+ "prompt_0": "{text}",
68
+ "prompt_1": "{text}\nTL;DR: ",
69
+ "prompt_2": "Given the text: {role1_output}\nReconstruct the summarized text to the detailed:",
70
+ "prompt_eval": "{text}\nTL;DR:",
71
+ "epsilon": 0.2,
72
+ "a2g_norm": true,
73
+ "vllm_sleep": true,
74
+ "lora": false,
75
+ "need_attn_mask": true,
76
+ "gamma": 0.95,
77
+ "trust_remote_code": true,
78
+ "test_glue": false,
79
+ "test_clm": false,
80
+ "causal_model": true,
81
+ "test_gen": true,
82
+ "log_rollout_txt": true,
83
+ "trunc_eval": 20000,
84
+ "buffer_max_size": 20000,
85
+ "trunc_evals": [
86
+ "xsum___20000"
87
+ ],
88
+ "use_deepspeed": false,
89
+ "zero_config": 2,
90
+ "log_interval": "5m",
91
+ "eval_interval": "2_000",
92
+ "checkpoint_interval": "2_000",
93
+ "lm_fraction": -1.0,
94
+ "push_to_hub": null,
95
+ "keep_eval_size": false,
96
+ "mixed_precision": "bf16",
97
+ "tune_metrics": [
98
+ "xsum/rouge1___1.0",
99
+ "xsum/rouge2___2.0",
100
+ "xsum/bertscore_f1___0.25"
101
+ ],
102
+ "base_path": "/common/home/users/m/mq.nguyen.2023/testcode/SAC_LM/module9_clmv3",
103
+ "script": "/common/home/users/m/mq.nguyen.2023/testcode/SAC_LM/module9_clmv3/execute/role_ablation/role_ablation.py",
104
+ "train_from_raw": true,
105
+ "world_size": 1,
106
+ "cpu_per_worker": 6,
107
+ "gpu_per_worker": 1,
108
+ "effective_batch_size": 16,
109
+ "text_col": "text",
110
+ "_scenario": "clamp_sft_ref",
111
+ "_dataset": "xsum"
112
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff