fxevangelinenyu commited on
Commit
a9f771e
·
verified ·
1 Parent(s): 8a7d3a5

upload rlrb_bs128_nobuf_seed2 (2/8)

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +4 -0
  2. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/added_tokens.json +24 -0
  3. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/chat_template.jinja +54 -0
  4. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/config.json +67 -0
  5. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/generation_config.json +6 -0
  6. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/merges.txt +0 -0
  7. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/model-00001-of-00002.safetensors +3 -0
  8. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/model-00002-of-00002.safetensors +3 -0
  9. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/model.safetensors.index.json +443 -0
  10. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/special_tokens_map.json +31 -0
  11. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/tokenizer.json +3 -0
  12. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/tokenizer_config.json +207 -0
  13. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/vocab.json +0 -0
  14. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/added_tokens.json +24 -0
  15. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/chat_template.jinja +54 -0
  16. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/config.json +67 -0
  17. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/eval_metrics.json +1151 -0
  18. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/generation_config.json +6 -0
  19. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/merges.txt +0 -0
  20. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/model-00001-of-00002.safetensors +3 -0
  21. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/model-00002-of-00002.safetensors +3 -0
  22. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/model.safetensors.index.json +443 -0
  23. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/special_tokens_map.json +31 -0
  24. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/tokenizer.json +3 -0
  25. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/tokenizer_config.json +207 -0
  26. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/vocab.json +0 -0
  27. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/added_tokens.json +24 -0
  28. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/chat_template.jinja +54 -0
  29. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/config.json +67 -0
  30. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/eval_metrics.json +1155 -0
  31. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/generation_config.json +6 -0
  32. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/merges.txt +0 -0
  33. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/model-00001-of-00002.safetensors +3 -0
  34. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/model-00002-of-00002.safetensors +3 -0
  35. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/model.safetensors.index.json +443 -0
  36. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/special_tokens_map.json +31 -0
  37. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/tokenizer.json +3 -0
  38. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/tokenizer_config.json +207 -0
  39. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/vocab.json +0 -0
  40. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/added_tokens.json +24 -0
  41. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/chat_template.jinja +54 -0
  42. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/config.json +67 -0
  43. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/eval_metrics.json +1159 -0
  44. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/generation_config.json +6 -0
  45. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/merges.txt +0 -0
  46. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/model-00001-of-00002.safetensors +3 -0
  47. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/model-00002-of-00002.safetensors +3 -0
  48. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/model.safetensors.index.json +443 -0
  49. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/special_tokens_map.json +31 -0
  50. rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/tokenizer.json +3 -0
.gitattributes CHANGED
@@ -132,3 +132,7 @@ rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_100/tokenizer.json filt
132
  rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
133
  rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1050/tokenizer.json filter=lfs diff=lfs merge=lfs -text
134
  rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1100/tokenizer.json filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
132
  rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
133
  rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1050/tokenizer.json filter=lfs diff=lfs merge=lfs -text
134
  rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1100/tokenizer.json filter=lfs diff=lfs merge=lfs -text
135
+ rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/tokenizer.json filter=lfs diff=lfs merge=lfs -text
136
+ rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
137
+ rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/tokenizer.json filter=lfs diff=lfs merge=lfs -text
138
+ rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/config.json ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "dtype": "bfloat16",
7
+ "eos_token_id": 151643,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 2048,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 11008,
12
+ "layer_types": [
13
+ "full_attention",
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention",
38
+ "full_attention",
39
+ "full_attention",
40
+ "full_attention",
41
+ "full_attention",
42
+ "full_attention",
43
+ "full_attention",
44
+ "full_attention",
45
+ "full_attention",
46
+ "full_attention",
47
+ "full_attention",
48
+ "full_attention"
49
+ ],
50
+ "max_position_embeddings": 32768,
51
+ "max_window_layers": 36,
52
+ "model_type": "qwen2",
53
+ "num_attention_heads": 16,
54
+ "num_hidden_layers": 36,
55
+ "num_key_value_heads": 2,
56
+ "pad_token_id": 151643,
57
+ "rms_norm_eps": 1e-06,
58
+ "rope_scaling": null,
59
+ "rope_theta": 1000000.0,
60
+ "sliding_window": null,
61
+ "tie_word_embeddings": true,
62
+ "transformers_version": "4.57.6",
63
+ "use_cache": true,
64
+ "use_mrope": false,
65
+ "use_sliding_window": false,
66
+ "vocab_size": 151936
67
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "eos_token_id": 151643,
4
+ "max_new_tokens": 2048,
5
+ "transformers_version": "4.57.6"
6
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cad6188577a9f56a7d588e9d2036bce939d67bbd0b413f94947a113574bbfa92
3
+ size 4973721584
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af7f5a1d86b1dfdd2f699b29025ff5c063622d9b580ff711497157e6e88d4992
3
+ size 1820535392
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/model.safetensors.index.json ADDED
@@ -0,0 +1,443 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_parameters": 3397103616,
4
+ "total_size": 6794207232
5
+ },
6
+ "weight_map": {
7
+ "lm_head.weight": "model-00001-of-00002.safetensors",
8
+ "model.embed_tokens.weight": "model-00001-of-00002.safetensors",
9
+ "model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
10
+ "model.layers.0.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
11
+ "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
12
+ "model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
13
+ "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
14
+ "model.layers.0.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
15
+ "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
16
+ "model.layers.0.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
17
+ "model.layers.0.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
18
+ "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
19
+ "model.layers.0.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
20
+ "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
21
+ "model.layers.1.input_layernorm.weight": "model-00001-of-00002.safetensors",
22
+ "model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
23
+ "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
24
+ "model.layers.1.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
25
+ "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
26
+ "model.layers.1.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
27
+ "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
28
+ "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
29
+ "model.layers.1.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
30
+ "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
31
+ "model.layers.1.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
32
+ "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
33
+ "model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors",
34
+ "model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
35
+ "model.layers.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
36
+ "model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
37
+ "model.layers.10.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
38
+ "model.layers.10.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
39
+ "model.layers.10.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
40
+ "model.layers.10.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
41
+ "model.layers.10.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
42
+ "model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
43
+ "model.layers.10.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
44
+ "model.layers.10.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
45
+ "model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
46
+ "model.layers.11.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
47
+ "model.layers.11.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
48
+ "model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
49
+ "model.layers.11.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
50
+ "model.layers.11.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
51
+ "model.layers.11.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
52
+ "model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
53
+ "model.layers.11.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
54
+ "model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
55
+ "model.layers.11.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
56
+ "model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
57
+ "model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
58
+ "model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
59
+ "model.layers.12.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
60
+ "model.layers.12.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
61
+ "model.layers.12.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
62
+ "model.layers.12.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
63
+ "model.layers.12.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
64
+ "model.layers.12.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
65
+ "model.layers.12.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
66
+ "model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
67
+ "model.layers.12.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
68
+ "model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
69
+ "model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
70
+ "model.layers.13.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
71
+ "model.layers.13.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
72
+ "model.layers.13.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
73
+ "model.layers.13.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
74
+ "model.layers.13.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
75
+ "model.layers.13.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
76
+ "model.layers.13.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
77
+ "model.layers.13.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
78
+ "model.layers.13.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
79
+ "model.layers.13.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
80
+ "model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
81
+ "model.layers.14.input_layernorm.weight": "model-00002-of-00002.safetensors",
82
+ "model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
83
+ "model.layers.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
84
+ "model.layers.14.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
85
+ "model.layers.14.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
86
+ "model.layers.14.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
87
+ "model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
88
+ "model.layers.14.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
89
+ "model.layers.14.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
90
+ "model.layers.14.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
91
+ "model.layers.14.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
92
+ "model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
93
+ "model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
94
+ "model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
95
+ "model.layers.15.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
96
+ "model.layers.15.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
97
+ "model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
98
+ "model.layers.15.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
99
+ "model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
100
+ "model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
101
+ "model.layers.15.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
102
+ "model.layers.15.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
103
+ "model.layers.15.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
104
+ "model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
105
+ "model.layers.16.input_layernorm.weight": "model-00002-of-00002.safetensors",
106
+ "model.layers.16.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
107
+ "model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
108
+ "model.layers.16.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
109
+ "model.layers.16.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
110
+ "model.layers.16.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
111
+ "model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
112
+ "model.layers.16.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
113
+ "model.layers.16.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
114
+ "model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
115
+ "model.layers.16.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
116
+ "model.layers.16.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
117
+ "model.layers.17.input_layernorm.weight": "model-00002-of-00002.safetensors",
118
+ "model.layers.17.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
119
+ "model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
120
+ "model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
121
+ "model.layers.17.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
122
+ "model.layers.17.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
123
+ "model.layers.17.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
124
+ "model.layers.17.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
125
+ "model.layers.17.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
126
+ "model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
127
+ "model.layers.17.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
128
+ "model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
129
+ "model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
130
+ "model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
131
+ "model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
132
+ "model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
133
+ "model.layers.18.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
134
+ "model.layers.18.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
135
+ "model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
136
+ "model.layers.18.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
137
+ "model.layers.18.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
138
+ "model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
139
+ "model.layers.18.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
140
+ "model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
141
+ "model.layers.19.input_layernorm.weight": "model-00002-of-00002.safetensors",
142
+ "model.layers.19.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
143
+ "model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
144
+ "model.layers.19.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
145
+ "model.layers.19.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
146
+ "model.layers.19.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
147
+ "model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
148
+ "model.layers.19.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
149
+ "model.layers.19.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
150
+ "model.layers.19.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
151
+ "model.layers.19.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
152
+ "model.layers.19.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
153
+ "model.layers.2.input_layernorm.weight": "model-00002-of-00002.safetensors",
154
+ "model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
155
+ "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
156
+ "model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
157
+ "model.layers.2.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
158
+ "model.layers.2.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
159
+ "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
160
+ "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
161
+ "model.layers.2.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
162
+ "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
163
+ "model.layers.2.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
164
+ "model.layers.2.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
165
+ "model.layers.20.input_layernorm.weight": "model-00002-of-00002.safetensors",
166
+ "model.layers.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
167
+ "model.layers.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
168
+ "model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
169
+ "model.layers.20.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
170
+ "model.layers.20.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
171
+ "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
172
+ "model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
173
+ "model.layers.20.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
174
+ "model.layers.20.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
175
+ "model.layers.20.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
176
+ "model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
177
+ "model.layers.21.input_layernorm.weight": "model-00001-of-00002.safetensors",
178
+ "model.layers.21.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
179
+ "model.layers.21.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
180
+ "model.layers.21.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
181
+ "model.layers.21.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
182
+ "model.layers.21.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
183
+ "model.layers.21.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
184
+ "model.layers.21.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
185
+ "model.layers.21.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
186
+ "model.layers.21.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
187
+ "model.layers.21.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
188
+ "model.layers.21.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
189
+ "model.layers.22.input_layernorm.weight": "model-00001-of-00002.safetensors",
190
+ "model.layers.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
191
+ "model.layers.22.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
192
+ "model.layers.22.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
193
+ "model.layers.22.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
194
+ "model.layers.22.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
195
+ "model.layers.22.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
196
+ "model.layers.22.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
197
+ "model.layers.22.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
198
+ "model.layers.22.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
199
+ "model.layers.22.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
200
+ "model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
201
+ "model.layers.23.input_layernorm.weight": "model-00002-of-00002.safetensors",
202
+ "model.layers.23.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
203
+ "model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
204
+ "model.layers.23.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
205
+ "model.layers.23.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
206
+ "model.layers.23.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
207
+ "model.layers.23.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
208
+ "model.layers.23.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
209
+ "model.layers.23.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
210
+ "model.layers.23.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
211
+ "model.layers.23.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
212
+ "model.layers.23.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
213
+ "model.layers.24.input_layernorm.weight": "model-00002-of-00002.safetensors",
214
+ "model.layers.24.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
215
+ "model.layers.24.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
216
+ "model.layers.24.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
217
+ "model.layers.24.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
218
+ "model.layers.24.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
219
+ "model.layers.24.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
220
+ "model.layers.24.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
221
+ "model.layers.24.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
222
+ "model.layers.24.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
223
+ "model.layers.24.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
224
+ "model.layers.24.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
225
+ "model.layers.25.input_layernorm.weight": "model-00002-of-00002.safetensors",
226
+ "model.layers.25.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
227
+ "model.layers.25.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
228
+ "model.layers.25.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
229
+ "model.layers.25.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
230
+ "model.layers.25.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
231
+ "model.layers.25.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
232
+ "model.layers.25.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
233
+ "model.layers.25.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
234
+ "model.layers.25.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
235
+ "model.layers.25.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
236
+ "model.layers.25.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
237
+ "model.layers.26.input_layernorm.weight": "model-00002-of-00002.safetensors",
238
+ "model.layers.26.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
239
+ "model.layers.26.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
240
+ "model.layers.26.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
241
+ "model.layers.26.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
242
+ "model.layers.26.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
243
+ "model.layers.26.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
244
+ "model.layers.26.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
245
+ "model.layers.26.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
246
+ "model.layers.26.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
247
+ "model.layers.26.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
248
+ "model.layers.26.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
249
+ "model.layers.27.input_layernorm.weight": "model-00001-of-00002.safetensors",
250
+ "model.layers.27.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
251
+ "model.layers.27.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
252
+ "model.layers.27.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
253
+ "model.layers.27.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
254
+ "model.layers.27.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
255
+ "model.layers.27.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
256
+ "model.layers.27.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
257
+ "model.layers.27.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
258
+ "model.layers.27.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
259
+ "model.layers.27.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
260
+ "model.layers.27.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
261
+ "model.layers.28.input_layernorm.weight": "model-00002-of-00002.safetensors",
262
+ "model.layers.28.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
263
+ "model.layers.28.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
264
+ "model.layers.28.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
265
+ "model.layers.28.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
266
+ "model.layers.28.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
267
+ "model.layers.28.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
268
+ "model.layers.28.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
269
+ "model.layers.28.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
270
+ "model.layers.28.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
271
+ "model.layers.28.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
272
+ "model.layers.28.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
273
+ "model.layers.29.input_layernorm.weight": "model-00001-of-00002.safetensors",
274
+ "model.layers.29.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
275
+ "model.layers.29.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
276
+ "model.layers.29.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
277
+ "model.layers.29.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
278
+ "model.layers.29.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
279
+ "model.layers.29.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
280
+ "model.layers.29.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
281
+ "model.layers.29.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
282
+ "model.layers.29.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
283
+ "model.layers.29.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
284
+ "model.layers.29.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
285
+ "model.layers.3.input_layernorm.weight": "model-00002-of-00002.safetensors",
286
+ "model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
287
+ "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
288
+ "model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
289
+ "model.layers.3.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
290
+ "model.layers.3.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
291
+ "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
292
+ "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
293
+ "model.layers.3.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
294
+ "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
295
+ "model.layers.3.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
296
+ "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
297
+ "model.layers.30.input_layernorm.weight": "model-00002-of-00002.safetensors",
298
+ "model.layers.30.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
299
+ "model.layers.30.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
300
+ "model.layers.30.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
301
+ "model.layers.30.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
302
+ "model.layers.30.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
303
+ "model.layers.30.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
304
+ "model.layers.30.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
305
+ "model.layers.30.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
306
+ "model.layers.30.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
307
+ "model.layers.30.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
308
+ "model.layers.30.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
309
+ "model.layers.31.input_layernorm.weight": "model-00002-of-00002.safetensors",
310
+ "model.layers.31.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
311
+ "model.layers.31.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
312
+ "model.layers.31.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
313
+ "model.layers.31.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
314
+ "model.layers.31.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
315
+ "model.layers.31.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
316
+ "model.layers.31.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
317
+ "model.layers.31.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
318
+ "model.layers.31.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
319
+ "model.layers.31.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
320
+ "model.layers.31.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
321
+ "model.layers.32.input_layernorm.weight": "model-00001-of-00002.safetensors",
322
+ "model.layers.32.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
323
+ "model.layers.32.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
324
+ "model.layers.32.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
325
+ "model.layers.32.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
326
+ "model.layers.32.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
327
+ "model.layers.32.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
328
+ "model.layers.32.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
329
+ "model.layers.32.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
330
+ "model.layers.32.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
331
+ "model.layers.32.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
332
+ "model.layers.32.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
333
+ "model.layers.33.input_layernorm.weight": "model-00002-of-00002.safetensors",
334
+ "model.layers.33.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
335
+ "model.layers.33.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
336
+ "model.layers.33.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
337
+ "model.layers.33.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
338
+ "model.layers.33.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
339
+ "model.layers.33.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
340
+ "model.layers.33.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
341
+ "model.layers.33.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
342
+ "model.layers.33.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
343
+ "model.layers.33.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
344
+ "model.layers.33.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
345
+ "model.layers.34.input_layernorm.weight": "model-00001-of-00002.safetensors",
346
+ "model.layers.34.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
347
+ "model.layers.34.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
348
+ "model.layers.34.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
349
+ "model.layers.34.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
350
+ "model.layers.34.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
351
+ "model.layers.34.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
352
+ "model.layers.34.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
353
+ "model.layers.34.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
354
+ "model.layers.34.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
355
+ "model.layers.34.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
356
+ "model.layers.34.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
357
+ "model.layers.35.input_layernorm.weight": "model-00001-of-00002.safetensors",
358
+ "model.layers.35.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
359
+ "model.layers.35.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
360
+ "model.layers.35.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
361
+ "model.layers.35.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
362
+ "model.layers.35.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
363
+ "model.layers.35.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
364
+ "model.layers.35.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
365
+ "model.layers.35.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
366
+ "model.layers.35.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
367
+ "model.layers.35.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
368
+ "model.layers.35.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
369
+ "model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors",
370
+ "model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
371
+ "model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
372
+ "model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
373
+ "model.layers.4.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
374
+ "model.layers.4.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
375
+ "model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
376
+ "model.layers.4.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
377
+ "model.layers.4.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
378
+ "model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
379
+ "model.layers.4.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
380
+ "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
381
+ "model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
382
+ "model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
383
+ "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
384
+ "model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
385
+ "model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
386
+ "model.layers.5.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
387
+ "model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
388
+ "model.layers.5.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
389
+ "model.layers.5.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
390
+ "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
391
+ "model.layers.5.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
392
+ "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
393
+ "model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
394
+ "model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
395
+ "model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
396
+ "model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
397
+ "model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
398
+ "model.layers.6.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
399
+ "model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
400
+ "model.layers.6.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
401
+ "model.layers.6.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
402
+ "model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
403
+ "model.layers.6.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
404
+ "model.layers.6.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
405
+ "model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
406
+ "model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
407
+ "model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
408
+ "model.layers.7.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
409
+ "model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
410
+ "model.layers.7.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
411
+ "model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
412
+ "model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
413
+ "model.layers.7.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
414
+ "model.layers.7.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
415
+ "model.layers.7.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
416
+ "model.layers.7.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
417
+ "model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
418
+ "model.layers.8.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
419
+ "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
420
+ "model.layers.8.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
421
+ "model.layers.8.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
422
+ "model.layers.8.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
423
+ "model.layers.8.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
424
+ "model.layers.8.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
425
+ "model.layers.8.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
426
+ "model.layers.8.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
427
+ "model.layers.8.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
428
+ "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
429
+ "model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
430
+ "model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
431
+ "model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
432
+ "model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
433
+ "model.layers.9.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
434
+ "model.layers.9.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
435
+ "model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
436
+ "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
437
+ "model.layers.9.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
438
+ "model.layers.9.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
439
+ "model.layers.9.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
440
+ "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
441
+ "model.norm.weight": "model-00001-of-00002.safetensors"
442
+ }
443
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|endoftext|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|endoftext|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 131072,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1150/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/config.json ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "dtype": "bfloat16",
7
+ "eos_token_id": 151643,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 2048,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 11008,
12
+ "layer_types": [
13
+ "full_attention",
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention",
38
+ "full_attention",
39
+ "full_attention",
40
+ "full_attention",
41
+ "full_attention",
42
+ "full_attention",
43
+ "full_attention",
44
+ "full_attention",
45
+ "full_attention",
46
+ "full_attention",
47
+ "full_attention",
48
+ "full_attention"
49
+ ],
50
+ "max_position_embeddings": 32768,
51
+ "max_window_layers": 36,
52
+ "model_type": "qwen2",
53
+ "num_attention_heads": 16,
54
+ "num_hidden_layers": 36,
55
+ "num_key_value_heads": 2,
56
+ "pad_token_id": 151643,
57
+ "rms_norm_eps": 1e-06,
58
+ "rope_scaling": null,
59
+ "rope_theta": 1000000.0,
60
+ "sliding_window": null,
61
+ "tie_word_embeddings": true,
62
+ "transformers_version": "4.57.6",
63
+ "use_cache": true,
64
+ "use_mrope": false,
65
+ "use_sliding_window": false,
66
+ "vocab_size": 151936
67
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/eval_metrics.json ADDED
@@ -0,0 +1,1151 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "global_step": 1200,
3
+ "ckpt_dir": "/mnt/mystorageoutput/projects/jy_intern/amlt-results/7213678591.49431-3a69baee-2340/qwen3b_kk_rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200",
4
+ "rollouts_dir": "/mnt/mystorageoutput/projects/jy_intern/amlt-results/7213678591.49431-3a69baee-2340/qwen3b_kk_rlrb_bs128_nobuf_seed2/rollouts/training",
5
+ "init_model_path": "Qwen/Qwen2.5-3B",
6
+ "prev_step": null,
7
+ "positive_score": 1.0,
8
+ "negative_score": 0.0,
9
+ "history_step_stride": 25,
10
+ "max_samples_per_class": 0,
11
+ "in_batch": {
12
+ "num_records": 1024,
13
+ "source_steps": [
14
+ 1200
15
+ ],
16
+ "num_source_steps": 1,
17
+ "num_positives_pre_cap": 687,
18
+ "num_negatives_pre_cap": 337,
19
+ "positive": {
20
+ "loss": 0.050488899432680805,
21
+ "token_entropy": 0.011468020771233295,
22
+ "num_supervised_tokens": 307636,
23
+ "kl_from_init": 3.757843129476044,
24
+ "kl_to_init": 0.32426493313794613,
25
+ "dead_units": {
26
+ "layer_0": {
27
+ "intermediate_size": 11008,
28
+ "dead_count": 0,
29
+ "dead_fraction": 0.0,
30
+ "mean_activation_freq": 0.9228337211400093,
31
+ "min_activation_freq": 0.7772107295635101
32
+ },
33
+ "layer_1": {
34
+ "intermediate_size": 11008,
35
+ "dead_count": 8851,
36
+ "dead_fraction": 0.8040515988372093,
37
+ "mean_activation_freq": 0.14600392125210948,
38
+ "min_activation_freq": 0.0
39
+ },
40
+ "layer_2": {
41
+ "intermediate_size": 11008,
42
+ "dead_count": 7516,
43
+ "dead_fraction": 0.6827761627906976,
44
+ "mean_activation_freq": 0.16621969904613185,
45
+ "min_activation_freq": 0.0
46
+ },
47
+ "layer_3": {
48
+ "intermediate_size": 11008,
49
+ "dead_count": 221,
50
+ "dead_fraction": 0.020076308139534885,
51
+ "mean_activation_freq": 0.3884043405306543,
52
+ "min_activation_freq": 0.0
53
+ },
54
+ "layer_4": {
55
+ "intermediate_size": 11008,
56
+ "dead_count": 23,
57
+ "dead_fraction": 0.0020893895348837207,
58
+ "mean_activation_freq": 0.5608594246566162,
59
+ "min_activation_freq": 0.0
60
+ },
61
+ "layer_5": {
62
+ "intermediate_size": 11008,
63
+ "dead_count": 22,
64
+ "dead_fraction": 0.001998546511627907,
65
+ "mean_activation_freq": 0.7210704696813806,
66
+ "min_activation_freq": 0.0
67
+ },
68
+ "layer_6": {
69
+ "intermediate_size": 11008,
70
+ "dead_count": 0,
71
+ "dead_fraction": 0.0,
72
+ "mean_activation_freq": 0.8755811584050406,
73
+ "min_activation_freq": 0.13125902040073334
74
+ },
75
+ "layer_7": {
76
+ "intermediate_size": 11008,
77
+ "dead_count": 0,
78
+ "dead_fraction": 0.0,
79
+ "mean_activation_freq": 0.8867085924816539,
80
+ "min_activation_freq": 0.04236500279551158
81
+ },
82
+ "layer_8": {
83
+ "intermediate_size": 11008,
84
+ "dead_count": 0,
85
+ "dead_fraction": 0.0,
86
+ "mean_activation_freq": 0.942254664412272,
87
+ "min_activation_freq": 0.07497822101444565
88
+ },
89
+ "layer_9": {
90
+ "intermediate_size": 11008,
91
+ "dead_count": 0,
92
+ "dead_fraction": 0.0,
93
+ "mean_activation_freq": 0.9503980863672472,
94
+ "min_activation_freq": 0.27314748598993616
95
+ },
96
+ "layer_10": {
97
+ "intermediate_size": 11008,
98
+ "dead_count": 0,
99
+ "dead_fraction": 0.0,
100
+ "mean_activation_freq": 0.9738114800526302,
101
+ "min_activation_freq": 0.6615643162698774
102
+ },
103
+ "layer_11": {
104
+ "intermediate_size": 11008,
105
+ "dead_count": 0,
106
+ "dead_fraction": 0.0,
107
+ "mean_activation_freq": 0.9852396080324997,
108
+ "min_activation_freq": 0.9328849679491348
109
+ },
110
+ "layer_12": {
111
+ "intermediate_size": 11008,
112
+ "dead_count": 0,
113
+ "dead_fraction": 0.0,
114
+ "mean_activation_freq": 0.9867702599395822,
115
+ "min_activation_freq": 0.9135049214006163
116
+ },
117
+ "layer_13": {
118
+ "intermediate_size": 11008,
119
+ "dead_count": 0,
120
+ "dead_fraction": 0.0,
121
+ "mean_activation_freq": 0.9805887317359092,
122
+ "min_activation_freq": 0.9110897294204839
123
+ },
124
+ "layer_14": {
125
+ "intermediate_size": 11008,
126
+ "dead_count": 0,
127
+ "dead_fraction": 0.0,
128
+ "mean_activation_freq": 0.9826279334209005,
129
+ "min_activation_freq": 0.8733373207296936
130
+ },
131
+ "layer_15": {
132
+ "intermediate_size": 11008,
133
+ "dead_count": 0,
134
+ "dead_fraction": 0.0,
135
+ "mean_activation_freq": 0.9802350511875141,
136
+ "min_activation_freq": 0.9033923207947054
137
+ },
138
+ "layer_16": {
139
+ "intermediate_size": 11008,
140
+ "dead_count": 0,
141
+ "dead_fraction": 0.0,
142
+ "mean_activation_freq": 0.9793264452551067,
143
+ "min_activation_freq": 0.9034410797175884
144
+ },
145
+ "layer_17": {
146
+ "intermediate_size": 11008,
147
+ "dead_count": 0,
148
+ "dead_fraction": 0.0,
149
+ "mean_activation_freq": 0.978973214734561,
150
+ "min_activation_freq": 0.9130173321717875
151
+ },
152
+ "layer_18": {
153
+ "intermediate_size": 11008,
154
+ "dead_count": 0,
155
+ "dead_fraction": 0.0,
156
+ "mean_activation_freq": 0.9796374162707242,
157
+ "min_activation_freq": 0.9189691713583586
158
+ },
159
+ "layer_19": {
160
+ "intermediate_size": 11008,
161
+ "dead_count": 0,
162
+ "dead_fraction": 0.0,
163
+ "mean_activation_freq": 0.9766303856970652,
164
+ "min_activation_freq": 0.9197818200730734
165
+ },
166
+ "layer_20": {
167
+ "intermediate_size": 11008,
168
+ "dead_count": 0,
169
+ "dead_fraction": 0.0,
170
+ "mean_activation_freq": 0.9793536046129873,
171
+ "min_activation_freq": 0.9129425684900337
172
+ },
173
+ "layer_21": {
174
+ "intermediate_size": 11008,
175
+ "dead_count": 0,
176
+ "dead_fraction": 0.0,
177
+ "mean_activation_freq": 0.979531047286668,
178
+ "min_activation_freq": 0.9179517351675357
179
+ },
180
+ "layer_22": {
181
+ "intermediate_size": 11008,
182
+ "dead_count": 0,
183
+ "dead_fraction": 0.0,
184
+ "mean_activation_freq": 0.9791617093717032,
185
+ "min_activation_freq": 0.9261042270735546
186
+ },
187
+ "layer_23": {
188
+ "intermediate_size": 11008,
189
+ "dead_count": 0,
190
+ "dead_fraction": 0.0,
191
+ "mean_activation_freq": 0.9794203253757575,
192
+ "min_activation_freq": 0.9357649949940839
193
+ },
194
+ "layer_24": {
195
+ "intermediate_size": 11008,
196
+ "dead_count": 0,
197
+ "dead_fraction": 0.0,
198
+ "mean_activation_freq": 0.9812896409570567,
199
+ "min_activation_freq": 0.9328297078365342
200
+ },
201
+ "layer_25": {
202
+ "intermediate_size": 11008,
203
+ "dead_count": 0,
204
+ "dead_fraction": 0.0,
205
+ "mean_activation_freq": 0.9833920432066613,
206
+ "min_activation_freq": 0.9462286598447517
207
+ },
208
+ "layer_26": {
209
+ "intermediate_size": 11008,
210
+ "dead_count": 0,
211
+ "dead_fraction": 0.0,
212
+ "mean_activation_freq": 0.9855332287618228,
213
+ "min_activation_freq": 0.9528826275208364
214
+ },
215
+ "layer_27": {
216
+ "intermediate_size": 11008,
217
+ "dead_count": 0,
218
+ "dead_fraction": 0.0,
219
+ "mean_activation_freq": 0.9867632830314489,
220
+ "min_activation_freq": 0.9381444304307689
221
+ },
222
+ "layer_28": {
223
+ "intermediate_size": 11008,
224
+ "dead_count": 0,
225
+ "dead_fraction": 0.0,
226
+ "mean_activation_freq": 0.9870280582749259,
227
+ "min_activation_freq": 0.23726416934298977
228
+ },
229
+ "layer_29": {
230
+ "intermediate_size": 11008,
231
+ "dead_count": 0,
232
+ "dead_fraction": 0.0,
233
+ "mean_activation_freq": 0.9883582703174651,
234
+ "min_activation_freq": 0.9584216411603323
235
+ },
236
+ "layer_30": {
237
+ "intermediate_size": 11008,
238
+ "dead_count": 0,
239
+ "dead_fraction": 0.0,
240
+ "mean_activation_freq": 0.9899633852971474,
241
+ "min_activation_freq": 0.5927557242975465
242
+ },
243
+ "layer_31": {
244
+ "intermediate_size": 11008,
245
+ "dead_count": 0,
246
+ "dead_fraction": 0.0,
247
+ "mean_activation_freq": 0.9907879340593021,
248
+ "min_activation_freq": 0.9377056001248228
249
+ },
250
+ "layer_32": {
251
+ "intermediate_size": 11008,
252
+ "dead_count": 0,
253
+ "dead_fraction": 0.0,
254
+ "mean_activation_freq": 0.9889488801932222,
255
+ "min_activation_freq": 0.9126402631681598
256
+ },
257
+ "layer_33": {
258
+ "intermediate_size": 11008,
259
+ "dead_count": 0,
260
+ "dead_fraction": 0.0,
261
+ "mean_activation_freq": 0.9865448836893692,
262
+ "min_activation_freq": 0.9436639405011118
263
+ },
264
+ "layer_34": {
265
+ "intermediate_size": 11008,
266
+ "dead_count": 0,
267
+ "dead_fraction": 0.0,
268
+ "mean_activation_freq": 0.9879019772182627,
269
+ "min_activation_freq": 0.8586381307779324
270
+ },
271
+ "layer_35": {
272
+ "intermediate_size": 11008,
273
+ "dead_count": 0,
274
+ "dead_fraction": 0.0,
275
+ "mean_activation_freq": 0.9910090675272717,
276
+ "min_activation_freq": 0.006514192097153779
277
+ },
278
+ "_overall": {
279
+ "dead_count": 16633,
280
+ "intermediate_total": 396288,
281
+ "dead_fraction": 0.041972000161498706
282
+ }
283
+ },
284
+ "size": 687,
285
+ "size_pre_filter": 687,
286
+ "skipped_long": 0
287
+ },
288
+ "negative": {
289
+ "loss": 0.051318609615137116,
290
+ "token_entropy": 0.013828090581555343,
291
+ "num_supervised_tokens": 182945,
292
+ "kl_from_init": 3.546219496200617,
293
+ "kl_to_init": 0.32233408307816747,
294
+ "dead_units": {
295
+ "layer_0": {
296
+ "intermediate_size": 11008,
297
+ "dead_count": 0,
298
+ "dead_fraction": 0.0,
299
+ "mean_activation_freq": 0.9230285050405922,
300
+ "min_activation_freq": 0.7765011342206674
301
+ },
302
+ "layer_1": {
303
+ "intermediate_size": 11008,
304
+ "dead_count": 8847,
305
+ "dead_fraction": 0.803688226744186,
306
+ "mean_activation_freq": 0.14602412246866034,
307
+ "min_activation_freq": 0.0
308
+ },
309
+ "layer_2": {
310
+ "intermediate_size": 11008,
311
+ "dead_count": 7331,
312
+ "dead_fraction": 0.6659702034883721,
313
+ "mean_activation_freq": 0.1661502568482267,
314
+ "min_activation_freq": 0.0
315
+ },
316
+ "layer_3": {
317
+ "intermediate_size": 11008,
318
+ "dead_count": 364,
319
+ "dead_fraction": 0.03306686046511628,
320
+ "mean_activation_freq": 0.3861732394950318,
321
+ "min_activation_freq": 0.0
322
+ },
323
+ "layer_4": {
324
+ "intermediate_size": 11008,
325
+ "dead_count": 26,
326
+ "dead_fraction": 0.0023619186046511626,
327
+ "mean_activation_freq": 0.5619188484617311,
328
+ "min_activation_freq": 0.0
329
+ },
330
+ "layer_5": {
331
+ "intermediate_size": 11008,
332
+ "dead_count": 26,
333
+ "dead_fraction": 0.0023619186046511626,
334
+ "mean_activation_freq": 0.7225178758333454,
335
+ "min_activation_freq": 0.0
336
+ },
337
+ "layer_6": {
338
+ "intermediate_size": 11008,
339
+ "dead_count": 0,
340
+ "dead_fraction": 0.0,
341
+ "mean_activation_freq": 0.8759900774759474,
342
+ "min_activation_freq": 0.1342643964032906
343
+ },
344
+ "layer_7": {
345
+ "intermediate_size": 11008,
346
+ "dead_count": 0,
347
+ "dead_fraction": 0.0,
348
+ "mean_activation_freq": 0.8896248532965493,
349
+ "min_activation_freq": 0.04996583672688513
350
+ },
351
+ "layer_8": {
352
+ "intermediate_size": 11008,
353
+ "dead_count": 0,
354
+ "dead_fraction": 0.0,
355
+ "mean_activation_freq": 0.9425454263282521,
356
+ "min_activation_freq": 0.07149689797480116
357
+ },
358
+ "layer_9": {
359
+ "intermediate_size": 11008,
360
+ "dead_count": 0,
361
+ "dead_fraction": 0.0,
362
+ "mean_activation_freq": 0.9511787396826916,
363
+ "min_activation_freq": 0.27865752001967803
364
+ },
365
+ "layer_10": {
366
+ "intermediate_size": 11008,
367
+ "dead_count": 0,
368
+ "dead_fraction": 0.0,
369
+ "mean_activation_freq": 0.9740915220977584,
370
+ "min_activation_freq": 0.6625105906146656
371
+ },
372
+ "layer_11": {
373
+ "intermediate_size": 11008,
374
+ "dead_count": 0,
375
+ "dead_fraction": 0.0,
376
+ "mean_activation_freq": 0.985233843334062,
377
+ "min_activation_freq": 0.9337888436415316
378
+ },
379
+ "layer_12": {
380
+ "intermediate_size": 11008,
381
+ "dead_count": 0,
382
+ "dead_fraction": 0.0,
383
+ "mean_activation_freq": 0.9867641797048546,
384
+ "min_activation_freq": 0.9097542977397578
385
+ },
386
+ "layer_13": {
387
+ "intermediate_size": 11008,
388
+ "dead_count": 0,
389
+ "dead_fraction": 0.0,
390
+ "mean_activation_freq": 0.9804354070426871,
391
+ "min_activation_freq": 0.9117494328896663
392
+ },
393
+ "layer_14": {
394
+ "intermediate_size": 11008,
395
+ "dead_count": 0,
396
+ "dead_fraction": 0.0,
397
+ "mean_activation_freq": 0.9825452106229347,
398
+ "min_activation_freq": 0.8690863374238158
399
+ },
400
+ "layer_15": {
401
+ "intermediate_size": 11008,
402
+ "dead_count": 0,
403
+ "dead_fraction": 0.0,
404
+ "mean_activation_freq": 0.9801415701209919,
405
+ "min_activation_freq": 0.8950886878570061
406
+ },
407
+ "layer_16": {
408
+ "intermediate_size": 11008,
409
+ "dead_count": 0,
410
+ "dead_fraction": 0.0,
411
+ "mean_activation_freq": 0.9792787448786869,
412
+ "min_activation_freq": 0.9026975320451501
413
+ },
414
+ "layer_17": {
415
+ "intermediate_size": 11008,
416
+ "dead_count": 0,
417
+ "dead_fraction": 0.0,
418
+ "mean_activation_freq": 0.9789626819670989,
419
+ "min_activation_freq": 0.9159091530241329
420
+ },
421
+ "layer_18": {
422
+ "intermediate_size": 11008,
423
+ "dead_count": 0,
424
+ "dead_fraction": 0.0,
425
+ "mean_activation_freq": 0.9796611649827085,
426
+ "min_activation_freq": 0.9184399682964824
427
+ },
428
+ "layer_19": {
429
+ "intermediate_size": 11008,
430
+ "dead_count": 0,
431
+ "dead_fraction": 0.0,
432
+ "mean_activation_freq": 0.976649334300816,
433
+ "min_activation_freq": 0.919456667304381
434
+ },
435
+ "layer_20": {
436
+ "intermediate_size": 11008,
437
+ "dead_count": 0,
438
+ "dead_fraction": 0.0,
439
+ "mean_activation_freq": 0.9793600678689173,
440
+ "min_activation_freq": 0.9163191123015113
441
+ },
442
+ "layer_21": {
443
+ "intermediate_size": 11008,
444
+ "dead_count": 0,
445
+ "dead_fraction": 0.0,
446
+ "mean_activation_freq": 0.9793450280837994,
447
+ "min_activation_freq": 0.9199704829320287
448
+ },
449
+ "layer_22": {
450
+ "intermediate_size": 11008,
451
+ "dead_count": 0,
452
+ "dead_fraction": 0.0,
453
+ "mean_activation_freq": 0.9788177984058624,
454
+ "min_activation_freq": 0.9256989805679302
455
+ },
456
+ "layer_23": {
457
+ "intermediate_size": 11008,
458
+ "dead_count": 0,
459
+ "dead_fraction": 0.0,
460
+ "mean_activation_freq": 0.9789745368214935,
461
+ "min_activation_freq": 0.9358659706469157
462
+ },
463
+ "layer_24": {
464
+ "intermediate_size": 11008,
465
+ "dead_count": 0,
466
+ "dead_fraction": 0.0,
467
+ "mean_activation_freq": 0.981032524448986,
468
+ "min_activation_freq": 0.9299024296919839
469
+ },
470
+ "layer_25": {
471
+ "intermediate_size": 11008,
472
+ "dead_count": 0,
473
+ "dead_fraction": 0.0,
474
+ "mean_activation_freq": 0.9830960164352355,
475
+ "min_activation_freq": 0.9427423542594768
476
+ },
477
+ "layer_26": {
478
+ "intermediate_size": 11008,
479
+ "dead_count": 0,
480
+ "dead_fraction": 0.0,
481
+ "mean_activation_freq": 0.9853047634090053,
482
+ "min_activation_freq": 0.9515974746508513
483
+ },
484
+ "layer_27": {
485
+ "intermediate_size": 11008,
486
+ "dead_count": 0,
487
+ "dead_fraction": 0.0,
488
+ "mean_activation_freq": 0.9865895294056797,
489
+ "min_activation_freq": 0.9377463171991581
490
+ },
491
+ "layer_28": {
492
+ "intermediate_size": 11008,
493
+ "dead_count": 0,
494
+ "dead_fraction": 0.0,
495
+ "mean_activation_freq": 0.9869042198276328,
496
+ "min_activation_freq": 0.24365246385525702
497
+ },
498
+ "layer_29": {
499
+ "intermediate_size": 11008,
500
+ "dead_count": 0,
501
+ "dead_fraction": 0.0,
502
+ "mean_activation_freq": 0.988245411832696,
503
+ "min_activation_freq": 0.9583700019131433
504
+ },
505
+ "layer_30": {
506
+ "intermediate_size": 11008,
507
+ "dead_count": 0,
508
+ "dead_fraction": 0.0,
509
+ "mean_activation_freq": 0.9898774097620836,
510
+ "min_activation_freq": 0.6105769493563639
511
+ },
512
+ "layer_31": {
513
+ "intermediate_size": 11008,
514
+ "dead_count": 0,
515
+ "dead_fraction": 0.0,
516
+ "mean_activation_freq": 0.9907216766007637,
517
+ "min_activation_freq": 0.9356090628330919
518
+ },
519
+ "layer_32": {
520
+ "intermediate_size": 11008,
521
+ "dead_count": 0,
522
+ "dead_fraction": 0.0,
523
+ "mean_activation_freq": 0.9888569796083394,
524
+ "min_activation_freq": 0.909743365492361
525
+ },
526
+ "layer_33": {
527
+ "intermediate_size": 11008,
528
+ "dead_count": 0,
529
+ "dead_fraction": 0.0,
530
+ "mean_activation_freq": 0.9864302371860713,
531
+ "min_activation_freq": 0.9446828281724015
532
+ },
533
+ "layer_34": {
534
+ "intermediate_size": 11008,
535
+ "dead_count": 0,
536
+ "dead_fraction": 0.0,
537
+ "mean_activation_freq": 0.9877653865622023,
538
+ "min_activation_freq": 0.8680259094263303
539
+ },
540
+ "layer_35": {
541
+ "intermediate_size": 11008,
542
+ "dead_count": 0,
543
+ "dead_fraction": 0.0,
544
+ "mean_activation_freq": 0.9913264027837188,
545
+ "min_activation_freq": 0.006931044849544945
546
+ },
547
+ "_overall": {
548
+ "dead_count": 16594,
549
+ "intermediate_total": 396288,
550
+ "dead_fraction": 0.04187358688630491
551
+ }
552
+ },
553
+ "size": 337,
554
+ "size_pre_filter": 337,
555
+ "skipped_long": 0
556
+ }
557
+ },
558
+ "old_data": {
559
+ "num_records": 48128,
560
+ "source_steps": [
561
+ 25,
562
+ 50,
563
+ 75,
564
+ 100,
565
+ 125,
566
+ 150,
567
+ 175,
568
+ 200,
569
+ 225,
570
+ 250,
571
+ 275,
572
+ 300,
573
+ 325,
574
+ 350,
575
+ 375,
576
+ 400,
577
+ 425,
578
+ 450,
579
+ 475,
580
+ 500,
581
+ 525,
582
+ 550,
583
+ 575,
584
+ 600,
585
+ 625,
586
+ 650,
587
+ 675,
588
+ 700,
589
+ 725,
590
+ 750,
591
+ 775,
592
+ 800,
593
+ 825,
594
+ 850,
595
+ 875,
596
+ 900,
597
+ 925,
598
+ 950,
599
+ 975,
600
+ 1000,
601
+ 1025,
602
+ 1050,
603
+ 1075,
604
+ 1100,
605
+ 1125,
606
+ 1150,
607
+ 1175
608
+ ],
609
+ "num_source_steps": 47,
610
+ "num_positives_pre_cap": 23863,
611
+ "num_negatives_pre_cap": 24265,
612
+ "positive": {
613
+ "loss": 0.5386253571582282,
614
+ "token_entropy": 0.014840442146156856,
615
+ "num_supervised_tokens": 9584732,
616
+ "kl_from_init": 3.7081134201994725,
617
+ "kl_to_init": 0.3491640592088618,
618
+ "dead_units": {
619
+ "layer_0": {
620
+ "intermediate_size": 11008,
621
+ "dead_count": 0,
622
+ "dead_fraction": 0.0,
623
+ "mean_activation_freq": 0.9231468462840292,
624
+ "min_activation_freq": 0.7777760504936392
625
+ },
626
+ "layer_1": {
627
+ "intermediate_size": 11008,
628
+ "dead_count": 8727,
629
+ "dead_fraction": 0.7927870639534884,
630
+ "mean_activation_freq": 0.1464105006863708,
631
+ "min_activation_freq": 0.0
632
+ },
633
+ "layer_2": {
634
+ "intermediate_size": 11008,
635
+ "dead_count": 3144,
636
+ "dead_fraction": 0.2856104651162791,
637
+ "mean_activation_freq": 0.16644836273707933,
638
+ "min_activation_freq": 0.0
639
+ },
640
+ "layer_3": {
641
+ "intermediate_size": 11008,
642
+ "dead_count": 39,
643
+ "dead_fraction": 0.003542877906976744,
644
+ "mean_activation_freq": 0.38815323211542835,
645
+ "min_activation_freq": 0.0
646
+ },
647
+ "layer_4": {
648
+ "intermediate_size": 11008,
649
+ "dead_count": 0,
650
+ "dead_fraction": 0.0,
651
+ "mean_activation_freq": 0.5575113028304067,
652
+ "min_activation_freq": 1.043325989709467e-07
653
+ },
654
+ "layer_5": {
655
+ "intermediate_size": 11008,
656
+ "dead_count": 0,
657
+ "dead_fraction": 0.0,
658
+ "mean_activation_freq": 0.7192566550803232,
659
+ "min_activation_freq": 2.2953171773608274e-06
660
+ },
661
+ "layer_6": {
662
+ "intermediate_size": 11008,
663
+ "dead_count": 0,
664
+ "dead_fraction": 0.0,
665
+ "mean_activation_freq": 0.8735564705099961,
666
+ "min_activation_freq": 0.126714758430387
667
+ },
668
+ "layer_7": {
669
+ "intermediate_size": 11008,
670
+ "dead_count": 0,
671
+ "dead_fraction": 0.0,
672
+ "mean_activation_freq": 0.88378483290686,
673
+ "min_activation_freq": 0.04364201315175009
674
+ },
675
+ "layer_8": {
676
+ "intermediate_size": 11008,
677
+ "dead_count": 0,
678
+ "dead_fraction": 0.0,
679
+ "mean_activation_freq": 0.9421332554819605,
680
+ "min_activation_freq": 0.07762668794495245
681
+ },
682
+ "layer_9": {
683
+ "intermediate_size": 11008,
684
+ "dead_count": 0,
685
+ "dead_fraction": 0.0,
686
+ "mean_activation_freq": 0.9504216159730168,
687
+ "min_activation_freq": 0.26264719764725813
688
+ },
689
+ "layer_10": {
690
+ "intermediate_size": 11008,
691
+ "dead_count": 0,
692
+ "dead_fraction": 0.0,
693
+ "mean_activation_freq": 0.9737855259486595,
694
+ "min_activation_freq": 0.6730875730276026
695
+ },
696
+ "layer_11": {
697
+ "intermediate_size": 11008,
698
+ "dead_count": 0,
699
+ "dead_fraction": 0.0,
700
+ "mean_activation_freq": 0.9852487892601477,
701
+ "min_activation_freq": 0.9343571630380484
702
+ },
703
+ "layer_12": {
704
+ "intermediate_size": 11008,
705
+ "dead_count": 0,
706
+ "dead_fraction": 0.0,
707
+ "mean_activation_freq": 0.9867846220696334,
708
+ "min_activation_freq": 0.9131508319690107
709
+ },
710
+ "layer_13": {
711
+ "intermediate_size": 11008,
712
+ "dead_count": 0,
713
+ "dead_fraction": 0.0,
714
+ "mean_activation_freq": 0.9807915201920635,
715
+ "min_activation_freq": 0.911102783051211
716
+ },
717
+ "layer_14": {
718
+ "intermediate_size": 11008,
719
+ "dead_count": 0,
720
+ "dead_fraction": 0.0,
721
+ "mean_activation_freq": 0.9827105949669787,
722
+ "min_activation_freq": 0.8776899552329683
723
+ },
724
+ "layer_15": {
725
+ "intermediate_size": 11008,
726
+ "dead_count": 0,
727
+ "dead_fraction": 0.0,
728
+ "mean_activation_freq": 0.9803994943563203,
729
+ "min_activation_freq": 0.9093039847123529
730
+ },
731
+ "layer_16": {
732
+ "intermediate_size": 11008,
733
+ "dead_count": 0,
734
+ "dead_fraction": 0.0,
735
+ "mean_activation_freq": 0.979459026528346,
736
+ "min_activation_freq": 0.9063116214412672
737
+ },
738
+ "layer_17": {
739
+ "intermediate_size": 11008,
740
+ "dead_count": 0,
741
+ "dead_fraction": 0.0,
742
+ "mean_activation_freq": 0.9790153988487797,
743
+ "min_activation_freq": 0.9154451058203817
744
+ },
745
+ "layer_18": {
746
+ "intermediate_size": 11008,
747
+ "dead_count": 0,
748
+ "dead_fraction": 0.0,
749
+ "mean_activation_freq": 0.9796628789681333,
750
+ "min_activation_freq": 0.9194975926296113
751
+ },
752
+ "layer_19": {
753
+ "intermediate_size": 11008,
754
+ "dead_count": 0,
755
+ "dead_fraction": 0.0,
756
+ "mean_activation_freq": 0.9766375082766686,
757
+ "min_activation_freq": 0.9219256208728631
758
+ },
759
+ "layer_20": {
760
+ "intermediate_size": 11008,
761
+ "dead_count": 0,
762
+ "dead_fraction": 0.0,
763
+ "mean_activation_freq": 0.9792923851231845,
764
+ "min_activation_freq": 0.9095633555533946
765
+ },
766
+ "layer_21": {
767
+ "intermediate_size": 11008,
768
+ "dead_count": 0,
769
+ "dead_fraction": 0.0,
770
+ "mean_activation_freq": 0.9794989081287191,
771
+ "min_activation_freq": 0.9165742975390443
772
+ },
773
+ "layer_22": {
774
+ "intermediate_size": 11008,
775
+ "dead_count": 0,
776
+ "dead_fraction": 0.0,
777
+ "mean_activation_freq": 0.9792026711344661,
778
+ "min_activation_freq": 0.9287330099579205
779
+ },
780
+ "layer_23": {
781
+ "intermediate_size": 11008,
782
+ "dead_count": 0,
783
+ "dead_fraction": 0.0,
784
+ "mean_activation_freq": 0.9795398765206799,
785
+ "min_activation_freq": 0.9372542706462736
786
+ },
787
+ "layer_24": {
788
+ "intermediate_size": 11008,
789
+ "dead_count": 0,
790
+ "dead_fraction": 0.0,
791
+ "mean_activation_freq": 0.9813737992327064,
792
+ "min_activation_freq": 0.9326645752849427
793
+ },
794
+ "layer_25": {
795
+ "intermediate_size": 11008,
796
+ "dead_count": 0,
797
+ "dead_fraction": 0.0,
798
+ "mean_activation_freq": 0.9834857025444429,
799
+ "min_activation_freq": 0.9474955585612618
800
+ },
801
+ "layer_26": {
802
+ "intermediate_size": 11008,
803
+ "dead_count": 0,
804
+ "dead_fraction": 0.0,
805
+ "mean_activation_freq": 0.9855754652806725,
806
+ "min_activation_freq": 0.9544878250116956
807
+ },
808
+ "layer_27": {
809
+ "intermediate_size": 11008,
810
+ "dead_count": 0,
811
+ "dead_fraction": 0.0,
812
+ "mean_activation_freq": 0.9868321753385131,
813
+ "min_activation_freq": 0.9381103196208302
814
+ },
815
+ "layer_28": {
816
+ "intermediate_size": 11008,
817
+ "dead_count": 0,
818
+ "dead_fraction": 0.0,
819
+ "mean_activation_freq": 0.9870813627568041,
820
+ "min_activation_freq": 0.24966728334188162
821
+ },
822
+ "layer_29": {
823
+ "intermediate_size": 11008,
824
+ "dead_count": 0,
825
+ "dead_fraction": 0.0,
826
+ "mean_activation_freq": 0.9884213372327832,
827
+ "min_activation_freq": 0.9588407897059614
828
+ },
829
+ "layer_30": {
830
+ "intermediate_size": 11008,
831
+ "dead_count": 0,
832
+ "dead_fraction": 0.0,
833
+ "mean_activation_freq": 0.9900144471984648,
834
+ "min_activation_freq": 0.5881905722559587
835
+ },
836
+ "layer_31": {
837
+ "intermediate_size": 11008,
838
+ "dead_count": 0,
839
+ "dead_fraction": 0.0,
840
+ "mean_activation_freq": 0.9908404382810884,
841
+ "min_activation_freq": 0.943052554834084
842
+ },
843
+ "layer_32": {
844
+ "intermediate_size": 11008,
845
+ "dead_count": 0,
846
+ "dead_fraction": 0.0,
847
+ "mean_activation_freq": 0.9890244399342075,
848
+ "min_activation_freq": 0.9122455380077398
849
+ },
850
+ "layer_33": {
851
+ "intermediate_size": 11008,
852
+ "dead_count": 0,
853
+ "dead_fraction": 0.0,
854
+ "mean_activation_freq": 0.9866516346758967,
855
+ "min_activation_freq": 0.939378586693921
856
+ },
857
+ "layer_34": {
858
+ "intermediate_size": 11008,
859
+ "dead_count": 0,
860
+ "dead_fraction": 0.0,
861
+ "mean_activation_freq": 0.9879729071828283,
862
+ "min_activation_freq": 0.8534473368686781
863
+ },
864
+ "layer_35": {
865
+ "intermediate_size": 11008,
866
+ "dead_count": 0,
867
+ "dead_fraction": 0.0,
868
+ "mean_activation_freq": 0.9909239298936974,
869
+ "min_activation_freq": 0.0071927937056560365
870
+ },
871
+ "_overall": {
872
+ "dead_count": 11910,
873
+ "intermediate_total": 396288,
874
+ "dead_fraction": 0.03005390019379845
875
+ }
876
+ },
877
+ "size": 23863,
878
+ "size_pre_filter": 23863,
879
+ "skipped_long": 0
880
+ },
881
+ "negative": {
882
+ "loss": 1.076902231627211,
883
+ "token_entropy": 0.030755371810873178,
884
+ "num_supervised_tokens": 11017395,
885
+ "kl_from_init": 3.4474491615996086,
886
+ "kl_to_init": 0.3705633982437457,
887
+ "dead_units": {
888
+ "layer_0": {
889
+ "intermediate_size": 11008,
890
+ "dead_count": 0,
891
+ "dead_fraction": 0.0,
892
+ "mean_activation_freq": 0.923437525940084,
893
+ "min_activation_freq": 0.7801131755737177
894
+ },
895
+ "layer_1": {
896
+ "intermediate_size": 11008,
897
+ "dead_count": 4951,
898
+ "dead_fraction": 0.44976380813953487,
899
+ "mean_activation_freq": 0.14648202010459474,
900
+ "min_activation_freq": 0.0
901
+ },
902
+ "layer_2": {
903
+ "intermediate_size": 11008,
904
+ "dead_count": 844,
905
+ "dead_fraction": 0.07667151162790697,
906
+ "mean_activation_freq": 0.16646534940209295,
907
+ "min_activation_freq": 0.0
908
+ },
909
+ "layer_3": {
910
+ "intermediate_size": 11008,
911
+ "dead_count": 0,
912
+ "dead_fraction": 0.0,
913
+ "mean_activation_freq": 0.3881595823815878,
914
+ "min_activation_freq": 9.984211331262972e-07
915
+ },
916
+ "layer_4": {
917
+ "intermediate_size": 11008,
918
+ "dead_count": 0,
919
+ "dead_fraction": 0.0,
920
+ "mean_activation_freq": 0.5587003114855598,
921
+ "min_activation_freq": 3.449091187163572e-06
922
+ },
923
+ "layer_5": {
924
+ "intermediate_size": 11008,
925
+ "dead_count": 0,
926
+ "dead_fraction": 0.0,
927
+ "mean_activation_freq": 0.7193636749468839,
928
+ "min_activation_freq": 3.0587992896687466e-05
929
+ },
930
+ "layer_6": {
931
+ "intermediate_size": 11008,
932
+ "dead_count": 0,
933
+ "dead_fraction": 0.0,
934
+ "mean_activation_freq": 0.8728662525752616,
935
+ "min_activation_freq": 0.12341084258120909
936
+ },
937
+ "layer_7": {
938
+ "intermediate_size": 11008,
939
+ "dead_count": 0,
940
+ "dead_fraction": 0.0,
941
+ "mean_activation_freq": 0.885259146422148,
942
+ "min_activation_freq": 0.04671666941232478
943
+ },
944
+ "layer_8": {
945
+ "intermediate_size": 11008,
946
+ "dead_count": 0,
947
+ "dead_fraction": 0.0,
948
+ "mean_activation_freq": 0.9422569147906245,
949
+ "min_activation_freq": 0.07827440152595055
950
+ },
951
+ "layer_9": {
952
+ "intermediate_size": 11008,
953
+ "dead_count": 0,
954
+ "dead_fraction": 0.0,
955
+ "mean_activation_freq": 0.9508600848834613,
956
+ "min_activation_freq": 0.2719126435967849
957
+ },
958
+ "layer_10": {
959
+ "intermediate_size": 11008,
960
+ "dead_count": 0,
961
+ "dead_fraction": 0.0,
962
+ "mean_activation_freq": 0.9738768826158569,
963
+ "min_activation_freq": 0.682505074929237
964
+ },
965
+ "layer_11": {
966
+ "intermediate_size": 11008,
967
+ "dead_count": 0,
968
+ "dead_fraction": 0.0,
969
+ "mean_activation_freq": 0.9852282719669511,
970
+ "min_activation_freq": 0.933349398837021
971
+ },
972
+ "layer_12": {
973
+ "intermediate_size": 11008,
974
+ "dead_count": 0,
975
+ "dead_fraction": 0.0,
976
+ "mean_activation_freq": 0.9867767499745935,
977
+ "min_activation_freq": 0.9126703726243818
978
+ },
979
+ "layer_13": {
980
+ "intermediate_size": 11008,
981
+ "dead_count": 0,
982
+ "dead_fraction": 0.0,
983
+ "mean_activation_freq": 0.9807496914406401,
984
+ "min_activation_freq": 0.9109593510988759
985
+ },
986
+ "layer_14": {
987
+ "intermediate_size": 11008,
988
+ "dead_count": 0,
989
+ "dead_fraction": 0.0,
990
+ "mean_activation_freq": 0.9826600459462245,
991
+ "min_activation_freq": 0.8794617057843529
992
+ },
993
+ "layer_15": {
994
+ "intermediate_size": 11008,
995
+ "dead_count": 0,
996
+ "dead_fraction": 0.0,
997
+ "mean_activation_freq": 0.9803886201324689,
998
+ "min_activation_freq": 0.9106896866273743
999
+ },
1000
+ "layer_16": {
1001
+ "intermediate_size": 11008,
1002
+ "dead_count": 0,
1003
+ "dead_fraction": 0.0,
1004
+ "mean_activation_freq": 0.979489539601617,
1005
+ "min_activation_freq": 0.9058762983445724
1006
+ },
1007
+ "layer_17": {
1008
+ "intermediate_size": 11008,
1009
+ "dead_count": 0,
1010
+ "dead_fraction": 0.0,
1011
+ "mean_activation_freq": 0.9790273987755062,
1012
+ "min_activation_freq": 0.9180569454031556
1013
+ },
1014
+ "layer_18": {
1015
+ "intermediate_size": 11008,
1016
+ "dead_count": 0,
1017
+ "dead_fraction": 0.0,
1018
+ "mean_activation_freq": 0.979676893127709,
1019
+ "min_activation_freq": 0.9204669524874074
1020
+ },
1021
+ "layer_19": {
1022
+ "intermediate_size": 11008,
1023
+ "dead_count": 0,
1024
+ "dead_fraction": 0.0,
1025
+ "mean_activation_freq": 0.9766194958311405,
1026
+ "min_activation_freq": 0.9224074293424172
1027
+ },
1028
+ "layer_20": {
1029
+ "intermediate_size": 11008,
1030
+ "dead_count": 0,
1031
+ "dead_fraction": 0.0,
1032
+ "mean_activation_freq": 0.9792816641297297,
1033
+ "min_activation_freq": 0.9141790777220932
1034
+ },
1035
+ "layer_21": {
1036
+ "intermediate_size": 11008,
1037
+ "dead_count": 0,
1038
+ "dead_fraction": 0.0,
1039
+ "mean_activation_freq": 0.9793612607248742,
1040
+ "min_activation_freq": 0.9153197284839112
1041
+ },
1042
+ "layer_22": {
1043
+ "intermediate_size": 11008,
1044
+ "dead_count": 0,
1045
+ "dead_fraction": 0.0,
1046
+ "mean_activation_freq": 0.9789635559825257,
1047
+ "min_activation_freq": 0.9301335751327787
1048
+ },
1049
+ "layer_23": {
1050
+ "intermediate_size": 11008,
1051
+ "dead_count": 0,
1052
+ "dead_fraction": 0.0,
1053
+ "mean_activation_freq": 0.9791967209666602,
1054
+ "min_activation_freq": 0.9358174958781091
1055
+ },
1056
+ "layer_24": {
1057
+ "intermediate_size": 11008,
1058
+ "dead_count": 0,
1059
+ "dead_fraction": 0.0,
1060
+ "mean_activation_freq": 0.9811544091168404,
1061
+ "min_activation_freq": 0.9318090165597221
1062
+ },
1063
+ "layer_25": {
1064
+ "intermediate_size": 11008,
1065
+ "dead_count": 0,
1066
+ "dead_fraction": 0.0,
1067
+ "mean_activation_freq": 0.9832154502791711,
1068
+ "min_activation_freq": 0.9455778793444367
1069
+ },
1070
+ "layer_26": {
1071
+ "intermediate_size": 11008,
1072
+ "dead_count": 0,
1073
+ "dead_fraction": 0.0,
1074
+ "mean_activation_freq": 0.9853622267253549,
1075
+ "min_activation_freq": 0.9542616017670239
1076
+ },
1077
+ "layer_27": {
1078
+ "intermediate_size": 11008,
1079
+ "dead_count": 0,
1080
+ "dead_fraction": 0.0,
1081
+ "mean_activation_freq": 0.9866825251922311,
1082
+ "min_activation_freq": 0.9365440741663524
1083
+ },
1084
+ "layer_28": {
1085
+ "intermediate_size": 11008,
1086
+ "dead_count": 0,
1087
+ "dead_fraction": 0.0,
1088
+ "mean_activation_freq": 0.9869856979526915,
1089
+ "min_activation_freq": 0.2661092753777095
1090
+ },
1091
+ "layer_29": {
1092
+ "intermediate_size": 11008,
1093
+ "dead_count": 0,
1094
+ "dead_fraction": 0.0,
1095
+ "mean_activation_freq": 0.9883252348341783,
1096
+ "min_activation_freq": 0.9584582380862264
1097
+ },
1098
+ "layer_30": {
1099
+ "intermediate_size": 11008,
1100
+ "dead_count": 0,
1101
+ "dead_fraction": 0.0,
1102
+ "mean_activation_freq": 0.9899484955477568,
1103
+ "min_activation_freq": 0.6063376142908555
1104
+ },
1105
+ "layer_31": {
1106
+ "intermediate_size": 11008,
1107
+ "dead_count": 0,
1108
+ "dead_fraction": 0.0,
1109
+ "mean_activation_freq": 0.9908241500554879,
1110
+ "min_activation_freq": 0.9440133534288278
1111
+ },
1112
+ "layer_32": {
1113
+ "intermediate_size": 11008,
1114
+ "dead_count": 0,
1115
+ "dead_fraction": 0.0,
1116
+ "mean_activation_freq": 0.9890242743179556,
1117
+ "min_activation_freq": 0.9081575998682084
1118
+ },
1119
+ "layer_33": {
1120
+ "intermediate_size": 11008,
1121
+ "dead_count": 0,
1122
+ "dead_fraction": 0.0,
1123
+ "mean_activation_freq": 0.9867016157500121,
1124
+ "min_activation_freq": 0.941690208983158
1125
+ },
1126
+ "layer_34": {
1127
+ "intermediate_size": 11008,
1128
+ "dead_count": 0,
1129
+ "dead_fraction": 0.0,
1130
+ "mean_activation_freq": 0.9879564475796002,
1131
+ "min_activation_freq": 0.8720169332224178
1132
+ },
1133
+ "layer_35": {
1134
+ "intermediate_size": 11008,
1135
+ "dead_count": 0,
1136
+ "dead_fraction": 0.0,
1137
+ "mean_activation_freq": 0.991359853175022,
1138
+ "min_activation_freq": 0.014356932832125925
1139
+ },
1140
+ "_overall": {
1141
+ "dead_count": 5795,
1142
+ "intermediate_total": 396288,
1143
+ "dead_fraction": 0.014623203326873386
1144
+ }
1145
+ },
1146
+ "size": 24265,
1147
+ "size_pre_filter": 24265,
1148
+ "skipped_long": 0
1149
+ }
1150
+ }
1151
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "eos_token_id": 151643,
4
+ "max_new_tokens": 2048,
5
+ "transformers_version": "4.57.6"
6
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5f9ec2a03378515740af2e5692e3a953f701a57adc81bec07a68f3f2b189b668
3
+ size 4998455304
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:04389cfcadb5dbfa841818f23c4e96d9a7fc36c69d77defc5fd884747932bc9f
3
+ size 1795801744
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/model.safetensors.index.json ADDED
@@ -0,0 +1,443 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_parameters": 3397103616,
4
+ "total_size": 6794207232
5
+ },
6
+ "weight_map": {
7
+ "lm_head.weight": "model-00001-of-00002.safetensors",
8
+ "model.embed_tokens.weight": "model-00002-of-00002.safetensors",
9
+ "model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
10
+ "model.layers.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
11
+ "model.layers.0.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
12
+ "model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
13
+ "model.layers.0.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
14
+ "model.layers.0.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
15
+ "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
16
+ "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
17
+ "model.layers.0.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
18
+ "model.layers.0.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
19
+ "model.layers.0.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
20
+ "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
21
+ "model.layers.1.input_layernorm.weight": "model-00002-of-00002.safetensors",
22
+ "model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
23
+ "model.layers.1.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
24
+ "model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
25
+ "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
26
+ "model.layers.1.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
27
+ "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
28
+ "model.layers.1.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
29
+ "model.layers.1.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
30
+ "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
31
+ "model.layers.1.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
32
+ "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
33
+ "model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors",
34
+ "model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
35
+ "model.layers.10.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
36
+ "model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
37
+ "model.layers.10.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
38
+ "model.layers.10.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
39
+ "model.layers.10.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
40
+ "model.layers.10.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
41
+ "model.layers.10.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
42
+ "model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
43
+ "model.layers.10.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
44
+ "model.layers.10.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
45
+ "model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
46
+ "model.layers.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
47
+ "model.layers.11.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
48
+ "model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
49
+ "model.layers.11.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
50
+ "model.layers.11.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
51
+ "model.layers.11.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
52
+ "model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
53
+ "model.layers.11.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
54
+ "model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
55
+ "model.layers.11.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
56
+ "model.layers.11.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
57
+ "model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
58
+ "model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
59
+ "model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
60
+ "model.layers.12.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
61
+ "model.layers.12.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
62
+ "model.layers.12.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
63
+ "model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
64
+ "model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
65
+ "model.layers.12.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
66
+ "model.layers.12.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
67
+ "model.layers.12.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
68
+ "model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
69
+ "model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
70
+ "model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
71
+ "model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
72
+ "model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
73
+ "model.layers.13.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
74
+ "model.layers.13.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
75
+ "model.layers.13.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
76
+ "model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
77
+ "model.layers.13.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
78
+ "model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
79
+ "model.layers.13.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
80
+ "model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
81
+ "model.layers.14.input_layernorm.weight": "model-00002-of-00002.safetensors",
82
+ "model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
83
+ "model.layers.14.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
84
+ "model.layers.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
85
+ "model.layers.14.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
86
+ "model.layers.14.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
87
+ "model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
88
+ "model.layers.14.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
89
+ "model.layers.14.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
90
+ "model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
91
+ "model.layers.14.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
92
+ "model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
93
+ "model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
94
+ "model.layers.15.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
95
+ "model.layers.15.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
96
+ "model.layers.15.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
97
+ "model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
98
+ "model.layers.15.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
99
+ "model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
100
+ "model.layers.15.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
101
+ "model.layers.15.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
102
+ "model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
103
+ "model.layers.15.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
104
+ "model.layers.15.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
105
+ "model.layers.16.input_layernorm.weight": "model-00001-of-00002.safetensors",
106
+ "model.layers.16.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
107
+ "model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
108
+ "model.layers.16.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
109
+ "model.layers.16.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
110
+ "model.layers.16.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
111
+ "model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
112
+ "model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
113
+ "model.layers.16.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
114
+ "model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
115
+ "model.layers.16.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
116
+ "model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
117
+ "model.layers.17.input_layernorm.weight": "model-00001-of-00002.safetensors",
118
+ "model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
119
+ "model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
120
+ "model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
121
+ "model.layers.17.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
122
+ "model.layers.17.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
123
+ "model.layers.17.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
124
+ "model.layers.17.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
125
+ "model.layers.17.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
126
+ "model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
127
+ "model.layers.17.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
128
+ "model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
129
+ "model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
130
+ "model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
131
+ "model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
132
+ "model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
133
+ "model.layers.18.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
134
+ "model.layers.18.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
135
+ "model.layers.18.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
136
+ "model.layers.18.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
137
+ "model.layers.18.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
138
+ "model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
139
+ "model.layers.18.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
140
+ "model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
141
+ "model.layers.19.input_layernorm.weight": "model-00001-of-00002.safetensors",
142
+ "model.layers.19.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
143
+ "model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
144
+ "model.layers.19.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
145
+ "model.layers.19.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
146
+ "model.layers.19.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
147
+ "model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
148
+ "model.layers.19.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
149
+ "model.layers.19.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
150
+ "model.layers.19.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
151
+ "model.layers.19.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
152
+ "model.layers.19.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
153
+ "model.layers.2.input_layernorm.weight": "model-00002-of-00002.safetensors",
154
+ "model.layers.2.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
155
+ "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
156
+ "model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
157
+ "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
158
+ "model.layers.2.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
159
+ "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
160
+ "model.layers.2.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
161
+ "model.layers.2.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
162
+ "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
163
+ "model.layers.2.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
164
+ "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
165
+ "model.layers.20.input_layernorm.weight": "model-00001-of-00002.safetensors",
166
+ "model.layers.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
167
+ "model.layers.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
168
+ "model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
169
+ "model.layers.20.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
170
+ "model.layers.20.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
171
+ "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
172
+ "model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
173
+ "model.layers.20.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
174
+ "model.layers.20.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
175
+ "model.layers.20.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
176
+ "model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
177
+ "model.layers.21.input_layernorm.weight": "model-00001-of-00002.safetensors",
178
+ "model.layers.21.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
179
+ "model.layers.21.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
180
+ "model.layers.21.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
181
+ "model.layers.21.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
182
+ "model.layers.21.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
183
+ "model.layers.21.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
184
+ "model.layers.21.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
185
+ "model.layers.21.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
186
+ "model.layers.21.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
187
+ "model.layers.21.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
188
+ "model.layers.21.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
189
+ "model.layers.22.input_layernorm.weight": "model-00001-of-00002.safetensors",
190
+ "model.layers.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
191
+ "model.layers.22.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
192
+ "model.layers.22.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
193
+ "model.layers.22.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
194
+ "model.layers.22.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
195
+ "model.layers.22.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
196
+ "model.layers.22.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
197
+ "model.layers.22.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
198
+ "model.layers.22.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
199
+ "model.layers.22.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
200
+ "model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
201
+ "model.layers.23.input_layernorm.weight": "model-00001-of-00002.safetensors",
202
+ "model.layers.23.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
203
+ "model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
204
+ "model.layers.23.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
205
+ "model.layers.23.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
206
+ "model.layers.23.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
207
+ "model.layers.23.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
208
+ "model.layers.23.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
209
+ "model.layers.23.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
210
+ "model.layers.23.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
211
+ "model.layers.23.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
212
+ "model.layers.23.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
213
+ "model.layers.24.input_layernorm.weight": "model-00001-of-00002.safetensors",
214
+ "model.layers.24.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
215
+ "model.layers.24.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
216
+ "model.layers.24.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
217
+ "model.layers.24.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
218
+ "model.layers.24.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
219
+ "model.layers.24.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
220
+ "model.layers.24.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
221
+ "model.layers.24.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
222
+ "model.layers.24.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
223
+ "model.layers.24.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
224
+ "model.layers.24.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
225
+ "model.layers.25.input_layernorm.weight": "model-00001-of-00002.safetensors",
226
+ "model.layers.25.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
227
+ "model.layers.25.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
228
+ "model.layers.25.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
229
+ "model.layers.25.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
230
+ "model.layers.25.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
231
+ "model.layers.25.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
232
+ "model.layers.25.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
233
+ "model.layers.25.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
234
+ "model.layers.25.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
235
+ "model.layers.25.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
236
+ "model.layers.25.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
237
+ "model.layers.26.input_layernorm.weight": "model-00001-of-00002.safetensors",
238
+ "model.layers.26.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
239
+ "model.layers.26.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
240
+ "model.layers.26.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
241
+ "model.layers.26.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
242
+ "model.layers.26.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
243
+ "model.layers.26.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
244
+ "model.layers.26.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
245
+ "model.layers.26.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
246
+ "model.layers.26.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
247
+ "model.layers.26.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
248
+ "model.layers.26.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
249
+ "model.layers.27.input_layernorm.weight": "model-00001-of-00002.safetensors",
250
+ "model.layers.27.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
251
+ "model.layers.27.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
252
+ "model.layers.27.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
253
+ "model.layers.27.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
254
+ "model.layers.27.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
255
+ "model.layers.27.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
256
+ "model.layers.27.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
257
+ "model.layers.27.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
258
+ "model.layers.27.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
259
+ "model.layers.27.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
260
+ "model.layers.27.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
261
+ "model.layers.28.input_layernorm.weight": "model-00001-of-00002.safetensors",
262
+ "model.layers.28.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
263
+ "model.layers.28.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
264
+ "model.layers.28.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
265
+ "model.layers.28.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
266
+ "model.layers.28.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
267
+ "model.layers.28.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
268
+ "model.layers.28.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
269
+ "model.layers.28.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
270
+ "model.layers.28.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
271
+ "model.layers.28.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
272
+ "model.layers.28.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
273
+ "model.layers.29.input_layernorm.weight": "model-00001-of-00002.safetensors",
274
+ "model.layers.29.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
275
+ "model.layers.29.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
276
+ "model.layers.29.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
277
+ "model.layers.29.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
278
+ "model.layers.29.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
279
+ "model.layers.29.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
280
+ "model.layers.29.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
281
+ "model.layers.29.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
282
+ "model.layers.29.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
283
+ "model.layers.29.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
284
+ "model.layers.29.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
285
+ "model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors",
286
+ "model.layers.3.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
287
+ "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
288
+ "model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
289
+ "model.layers.3.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
290
+ "model.layers.3.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
291
+ "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
292
+ "model.layers.3.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
293
+ "model.layers.3.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
294
+ "model.layers.3.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
295
+ "model.layers.3.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
296
+ "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
297
+ "model.layers.30.input_layernorm.weight": "model-00002-of-00002.safetensors",
298
+ "model.layers.30.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
299
+ "model.layers.30.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
300
+ "model.layers.30.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
301
+ "model.layers.30.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
302
+ "model.layers.30.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
303
+ "model.layers.30.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
304
+ "model.layers.30.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
305
+ "model.layers.30.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
306
+ "model.layers.30.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
307
+ "model.layers.30.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
308
+ "model.layers.30.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
309
+ "model.layers.31.input_layernorm.weight": "model-00001-of-00002.safetensors",
310
+ "model.layers.31.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
311
+ "model.layers.31.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
312
+ "model.layers.31.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
313
+ "model.layers.31.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
314
+ "model.layers.31.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
315
+ "model.layers.31.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
316
+ "model.layers.31.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
317
+ "model.layers.31.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
318
+ "model.layers.31.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
319
+ "model.layers.31.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
320
+ "model.layers.31.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
321
+ "model.layers.32.input_layernorm.weight": "model-00001-of-00002.safetensors",
322
+ "model.layers.32.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
323
+ "model.layers.32.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
324
+ "model.layers.32.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
325
+ "model.layers.32.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
326
+ "model.layers.32.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
327
+ "model.layers.32.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
328
+ "model.layers.32.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
329
+ "model.layers.32.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
330
+ "model.layers.32.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
331
+ "model.layers.32.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
332
+ "model.layers.32.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
333
+ "model.layers.33.input_layernorm.weight": "model-00001-of-00002.safetensors",
334
+ "model.layers.33.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
335
+ "model.layers.33.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
336
+ "model.layers.33.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
337
+ "model.layers.33.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
338
+ "model.layers.33.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
339
+ "model.layers.33.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
340
+ "model.layers.33.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
341
+ "model.layers.33.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
342
+ "model.layers.33.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
343
+ "model.layers.33.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
344
+ "model.layers.33.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
345
+ "model.layers.34.input_layernorm.weight": "model-00001-of-00002.safetensors",
346
+ "model.layers.34.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
347
+ "model.layers.34.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
348
+ "model.layers.34.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
349
+ "model.layers.34.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
350
+ "model.layers.34.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
351
+ "model.layers.34.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
352
+ "model.layers.34.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
353
+ "model.layers.34.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
354
+ "model.layers.34.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
355
+ "model.layers.34.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
356
+ "model.layers.34.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
357
+ "model.layers.35.input_layernorm.weight": "model-00001-of-00002.safetensors",
358
+ "model.layers.35.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
359
+ "model.layers.35.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
360
+ "model.layers.35.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
361
+ "model.layers.35.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
362
+ "model.layers.35.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
363
+ "model.layers.35.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
364
+ "model.layers.35.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
365
+ "model.layers.35.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
366
+ "model.layers.35.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
367
+ "model.layers.35.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
368
+ "model.layers.35.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
369
+ "model.layers.4.input_layernorm.weight": "model-00002-of-00002.safetensors",
370
+ "model.layers.4.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
371
+ "model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
372
+ "model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
373
+ "model.layers.4.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
374
+ "model.layers.4.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
375
+ "model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
376
+ "model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
377
+ "model.layers.4.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
378
+ "model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
379
+ "model.layers.4.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
380
+ "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
381
+ "model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
382
+ "model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
383
+ "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
384
+ "model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
385
+ "model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
386
+ "model.layers.5.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
387
+ "model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
388
+ "model.layers.5.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
389
+ "model.layers.5.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
390
+ "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
391
+ "model.layers.5.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
392
+ "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
393
+ "model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
394
+ "model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
395
+ "model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
396
+ "model.layers.6.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
397
+ "model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
398
+ "model.layers.6.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
399
+ "model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
400
+ "model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
401
+ "model.layers.6.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
402
+ "model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
403
+ "model.layers.6.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
404
+ "model.layers.6.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
405
+ "model.layers.7.input_layernorm.weight": "model-00002-of-00002.safetensors",
406
+ "model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
407
+ "model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
408
+ "model.layers.7.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
409
+ "model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
410
+ "model.layers.7.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
411
+ "model.layers.7.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
412
+ "model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
413
+ "model.layers.7.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
414
+ "model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
415
+ "model.layers.7.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
416
+ "model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
417
+ "model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
418
+ "model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
419
+ "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
420
+ "model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
421
+ "model.layers.8.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
422
+ "model.layers.8.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
423
+ "model.layers.8.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
424
+ "model.layers.8.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
425
+ "model.layers.8.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
426
+ "model.layers.8.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
427
+ "model.layers.8.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
428
+ "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
429
+ "model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
430
+ "model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
431
+ "model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
432
+ "model.layers.9.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
433
+ "model.layers.9.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
434
+ "model.layers.9.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
435
+ "model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
436
+ "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
437
+ "model.layers.9.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
438
+ "model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
439
+ "model.layers.9.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
440
+ "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
441
+ "model.norm.weight": "model-00001-of-00002.safetensors"
442
+ }
443
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|endoftext|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|endoftext|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 131072,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1200/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/config.json ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "dtype": "bfloat16",
7
+ "eos_token_id": 151643,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 2048,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 11008,
12
+ "layer_types": [
13
+ "full_attention",
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention",
38
+ "full_attention",
39
+ "full_attention",
40
+ "full_attention",
41
+ "full_attention",
42
+ "full_attention",
43
+ "full_attention",
44
+ "full_attention",
45
+ "full_attention",
46
+ "full_attention",
47
+ "full_attention",
48
+ "full_attention"
49
+ ],
50
+ "max_position_embeddings": 32768,
51
+ "max_window_layers": 36,
52
+ "model_type": "qwen2",
53
+ "num_attention_heads": 16,
54
+ "num_hidden_layers": 36,
55
+ "num_key_value_heads": 2,
56
+ "pad_token_id": 151643,
57
+ "rms_norm_eps": 1e-06,
58
+ "rope_scaling": null,
59
+ "rope_theta": 1000000.0,
60
+ "sliding_window": null,
61
+ "tie_word_embeddings": true,
62
+ "transformers_version": "4.57.6",
63
+ "use_cache": true,
64
+ "use_mrope": false,
65
+ "use_sliding_window": false,
66
+ "vocab_size": 151936
67
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/eval_metrics.json ADDED
@@ -0,0 +1,1155 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "global_step": 1300,
3
+ "ckpt_dir": "/mnt/mystorageoutput/projects/jy_intern/amlt-results/7213678591.49431-3a69baee-2340/qwen3b_kk_rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300",
4
+ "rollouts_dir": "/mnt/mystorageoutput/projects/jy_intern/amlt-results/7213678591.49431-3a69baee-2340/qwen3b_kk_rlrb_bs128_nobuf_seed2/rollouts/training",
5
+ "init_model_path": "Qwen/Qwen2.5-3B",
6
+ "prev_step": null,
7
+ "positive_score": 1.0,
8
+ "negative_score": 0.0,
9
+ "history_step_stride": 25,
10
+ "max_samples_per_class": 0,
11
+ "in_batch": {
12
+ "num_records": 1024,
13
+ "source_steps": [
14
+ 1300
15
+ ],
16
+ "num_source_steps": 1,
17
+ "num_positives_pre_cap": 711,
18
+ "num_negatives_pre_cap": 313,
19
+ "positive": {
20
+ "loss": 0.05142938777293532,
21
+ "token_entropy": 0.012691659669387382,
22
+ "num_supervised_tokens": 352225,
23
+ "kl_from_init": 3.418892523436391,
24
+ "kl_to_init": 0.3135166419204754,
25
+ "dead_units": {
26
+ "layer_0": {
27
+ "intermediate_size": 11008,
28
+ "dead_count": 0,
29
+ "dead_fraction": 0.0,
30
+ "mean_activation_freq": 0.9232493653819489,
31
+ "min_activation_freq": 0.778641493363617
32
+ },
33
+ "layer_1": {
34
+ "intermediate_size": 11008,
35
+ "dead_count": 8835,
36
+ "dead_fraction": 0.8025981104651163,
37
+ "mean_activation_freq": 0.14591937162960714,
38
+ "min_activation_freq": 0.0
39
+ },
40
+ "layer_2": {
41
+ "intermediate_size": 11008,
42
+ "dead_count": 7577,
43
+ "dead_fraction": 0.6883175872093024,
44
+ "mean_activation_freq": 0.16619702308786172,
45
+ "min_activation_freq": 0.0
46
+ },
47
+ "layer_3": {
48
+ "intermediate_size": 11008,
49
+ "dead_count": 240,
50
+ "dead_fraction": 0.02180232558139535,
51
+ "mean_activation_freq": 0.38751640036058144,
52
+ "min_activation_freq": 0.0
53
+ },
54
+ "layer_4": {
55
+ "intermediate_size": 11008,
56
+ "dead_count": 16,
57
+ "dead_fraction": 0.0014534883720930232,
58
+ "mean_activation_freq": 0.5618245222542904,
59
+ "min_activation_freq": 0.0
60
+ },
61
+ "layer_5": {
62
+ "intermediate_size": 11008,
63
+ "dead_count": 20,
64
+ "dead_fraction": 0.001816860465116279,
65
+ "mean_activation_freq": 0.7225563171293125,
66
+ "min_activation_freq": 0.0
67
+ },
68
+ "layer_6": {
69
+ "intermediate_size": 11008,
70
+ "dead_count": 0,
71
+ "dead_fraction": 0.0,
72
+ "mean_activation_freq": 0.8754557414389751,
73
+ "min_activation_freq": 0.1345446802470012
74
+ },
75
+ "layer_7": {
76
+ "intermediate_size": 11008,
77
+ "dead_count": 0,
78
+ "dead_fraction": 0.0,
79
+ "mean_activation_freq": 0.8868553375695536,
80
+ "min_activation_freq": 0.04861665128823905
81
+ },
82
+ "layer_8": {
83
+ "intermediate_size": 11008,
84
+ "dead_count": 0,
85
+ "dead_fraction": 0.0,
86
+ "mean_activation_freq": 0.9421857508414118,
87
+ "min_activation_freq": 0.06926254524806587
88
+ },
89
+ "layer_9": {
90
+ "intermediate_size": 11008,
91
+ "dead_count": 0,
92
+ "dead_fraction": 0.0,
93
+ "mean_activation_freq": 0.9504666077836577,
94
+ "min_activation_freq": 0.271068209241252
95
+ },
96
+ "layer_10": {
97
+ "intermediate_size": 11008,
98
+ "dead_count": 0,
99
+ "dead_fraction": 0.0,
100
+ "mean_activation_freq": 0.97394656317934,
101
+ "min_activation_freq": 0.6606657676201292
102
+ },
103
+ "layer_11": {
104
+ "intermediate_size": 11008,
105
+ "dead_count": 0,
106
+ "dead_fraction": 0.0,
107
+ "mean_activation_freq": 0.9852589667718671,
108
+ "min_activation_freq": 0.9342437362481368
109
+ },
110
+ "layer_12": {
111
+ "intermediate_size": 11008,
112
+ "dead_count": 0,
113
+ "dead_fraction": 0.0,
114
+ "mean_activation_freq": 0.9867302446181004,
115
+ "min_activation_freq": 0.9121442259919086
116
+ },
117
+ "layer_13": {
118
+ "intermediate_size": 11008,
119
+ "dead_count": 0,
120
+ "dead_fraction": 0.0,
121
+ "mean_activation_freq": 0.9805039100993352,
122
+ "min_activation_freq": 0.9107530697707431
123
+ },
124
+ "layer_14": {
125
+ "intermediate_size": 11008,
126
+ "dead_count": 0,
127
+ "dead_fraction": 0.0,
128
+ "mean_activation_freq": 0.9827553609054235,
129
+ "min_activation_freq": 0.8656710909219959
130
+ },
131
+ "layer_15": {
132
+ "intermediate_size": 11008,
133
+ "dead_count": 0,
134
+ "dead_fraction": 0.0,
135
+ "mean_activation_freq": 0.9804264568309105,
136
+ "min_activation_freq": 0.9047029597558378
137
+ },
138
+ "layer_16": {
139
+ "intermediate_size": 11008,
140
+ "dead_count": 0,
141
+ "dead_fraction": 0.0,
142
+ "mean_activation_freq": 0.979438149989601,
143
+ "min_activation_freq": 0.9066903257860742
144
+ },
145
+ "layer_17": {
146
+ "intermediate_size": 11008,
147
+ "dead_count": 0,
148
+ "dead_fraction": 0.0,
149
+ "mean_activation_freq": 0.9790791820003897,
150
+ "min_activation_freq": 0.9111959684860529
151
+ },
152
+ "layer_18": {
153
+ "intermediate_size": 11008,
154
+ "dead_count": 0,
155
+ "dead_fraction": 0.0,
156
+ "mean_activation_freq": 0.9797813554343896,
157
+ "min_activation_freq": 0.9207608772801477
158
+ },
159
+ "layer_19": {
160
+ "intermediate_size": 11008,
161
+ "dead_count": 0,
162
+ "dead_fraction": 0.0,
163
+ "mean_activation_freq": 0.9768337585956882,
164
+ "min_activation_freq": 0.9180864504223153
165
+ },
166
+ "layer_20": {
167
+ "intermediate_size": 11008,
168
+ "dead_count": 0,
169
+ "dead_fraction": 0.0,
170
+ "mean_activation_freq": 0.979454566598633,
171
+ "min_activation_freq": 0.911633189012705
172
+ },
173
+ "layer_21": {
174
+ "intermediate_size": 11008,
175
+ "dead_count": 0,
176
+ "dead_fraction": 0.0,
177
+ "mean_activation_freq": 0.9795606429310678,
178
+ "min_activation_freq": 0.9182965434026545
179
+ },
180
+ "layer_22": {
181
+ "intermediate_size": 11008,
182
+ "dead_count": 0,
183
+ "dead_fraction": 0.0,
184
+ "mean_activation_freq": 0.9790515761409612,
185
+ "min_activation_freq": 0.9262346511462843
186
+ },
187
+ "layer_23": {
188
+ "intermediate_size": 11008,
189
+ "dead_count": 0,
190
+ "dead_fraction": 0.0,
191
+ "mean_activation_freq": 0.9792129240278163,
192
+ "min_activation_freq": 0.9368045993328128
193
+ },
194
+ "layer_24": {
195
+ "intermediate_size": 11008,
196
+ "dead_count": 0,
197
+ "dead_fraction": 0.0,
198
+ "mean_activation_freq": 0.9810340328179497,
199
+ "min_activation_freq": 0.9327532117254596
200
+ },
201
+ "layer_25": {
202
+ "intermediate_size": 11008,
203
+ "dead_count": 0,
204
+ "dead_fraction": 0.0,
205
+ "mean_activation_freq": 0.9830869128583739,
206
+ "min_activation_freq": 0.9426133863297608
207
+ },
208
+ "layer_26": {
209
+ "intermediate_size": 11008,
210
+ "dead_count": 0,
211
+ "dead_fraction": 0.0,
212
+ "mean_activation_freq": 0.9852788295998692,
213
+ "min_activation_freq": 0.9511079565618568
214
+ },
215
+ "layer_27": {
216
+ "intermediate_size": 11008,
217
+ "dead_count": 0,
218
+ "dead_fraction": 0.0,
219
+ "mean_activation_freq": 0.9866414308457694,
220
+ "min_activation_freq": 0.9394307615870537
221
+ },
222
+ "layer_28": {
223
+ "intermediate_size": 11008,
224
+ "dead_count": 0,
225
+ "dead_fraction": 0.0,
226
+ "mean_activation_freq": 0.9869614987550076,
227
+ "min_activation_freq": 0.24921286109730995
228
+ },
229
+ "layer_29": {
230
+ "intermediate_size": 11008,
231
+ "dead_count": 0,
232
+ "dead_fraction": 0.0,
233
+ "mean_activation_freq": 0.9883285288642633,
234
+ "min_activation_freq": 0.9567889843140038
235
+ },
236
+ "layer_30": {
237
+ "intermediate_size": 11008,
238
+ "dead_count": 0,
239
+ "dead_fraction": 0.0,
240
+ "mean_activation_freq": 0.9899693696075776,
241
+ "min_activation_freq": 0.6177897650649443
242
+ },
243
+ "layer_31": {
244
+ "intermediate_size": 11008,
245
+ "dead_count": 0,
246
+ "dead_fraction": 0.0,
247
+ "mean_activation_freq": 0.9907809252889025,
248
+ "min_activation_freq": 0.9378380296685357
249
+ },
250
+ "layer_32": {
251
+ "intermediate_size": 11008,
252
+ "dead_count": 0,
253
+ "dead_fraction": 0.0,
254
+ "mean_activation_freq": 0.9889304026768367,
255
+ "min_activation_freq": 0.9172744694442473
256
+ },
257
+ "layer_33": {
258
+ "intermediate_size": 11008,
259
+ "dead_count": 0,
260
+ "dead_fraction": 0.0,
261
+ "mean_activation_freq": 0.9865632868892438,
262
+ "min_activation_freq": 0.9348598197175101
263
+ },
264
+ "layer_34": {
265
+ "intermediate_size": 11008,
266
+ "dead_count": 0,
267
+ "dead_fraction": 0.0,
268
+ "mean_activation_freq": 0.9877815851307385,
269
+ "min_activation_freq": 0.8898544964156434
270
+ },
271
+ "layer_35": {
272
+ "intermediate_size": 11008,
273
+ "dead_count": 0,
274
+ "dead_fraction": 0.0,
275
+ "mean_activation_freq": 0.9915231545061545,
276
+ "min_activation_freq": 0.006969976577471787
277
+ },
278
+ "_overall": {
279
+ "dead_count": 16688,
280
+ "intermediate_total": 396288,
281
+ "dead_fraction": 0.04211078811369509
282
+ }
283
+ },
284
+ "size": 711,
285
+ "size_pre_filter": 711,
286
+ "skipped_long": 0
287
+ },
288
+ "negative": {
289
+ "loss": 0.04896152354399487,
290
+ "token_entropy": 0.015343843917676932,
291
+ "num_supervised_tokens": 195064,
292
+ "kl_from_init": 3.182585433348558,
293
+ "kl_to_init": 0.30884841104158645,
294
+ "dead_units": {
295
+ "layer_0": {
296
+ "intermediate_size": 11008,
297
+ "dead_count": 0,
298
+ "dead_fraction": 0.0,
299
+ "mean_activation_freq": 0.9236309075646848,
300
+ "min_activation_freq": 0.7782061272197843
301
+ },
302
+ "layer_1": {
303
+ "intermediate_size": 11008,
304
+ "dead_count": 8919,
305
+ "dead_fraction": 0.8102289244186046,
306
+ "mean_activation_freq": 0.14568002509790468,
307
+ "min_activation_freq": 0.0
308
+ },
309
+ "layer_2": {
310
+ "intermediate_size": 11008,
311
+ "dead_count": 7687,
312
+ "dead_fraction": 0.6983103197674418,
313
+ "mean_activation_freq": 0.1660920622526453,
314
+ "min_activation_freq": 0.0
315
+ },
316
+ "layer_3": {
317
+ "intermediate_size": 11008,
318
+ "dead_count": 118,
319
+ "dead_fraction": 0.010719476744186046,
320
+ "mean_activation_freq": 0.38835409160806733,
321
+ "min_activation_freq": 0.0
322
+ },
323
+ "layer_4": {
324
+ "intermediate_size": 11008,
325
+ "dead_count": 28,
326
+ "dead_fraction": 0.002543604651162791,
327
+ "mean_activation_freq": 0.5639313457810269,
328
+ "min_activation_freq": 0.0
329
+ },
330
+ "layer_5": {
331
+ "intermediate_size": 11008,
332
+ "dead_count": 26,
333
+ "dead_fraction": 0.0023619186046511626,
334
+ "mean_activation_freq": 0.7234873194793432,
335
+ "min_activation_freq": 0.0
336
+ },
337
+ "layer_6": {
338
+ "intermediate_size": 11008,
339
+ "dead_count": 0,
340
+ "dead_fraction": 0.0,
341
+ "mean_activation_freq": 0.8745722865092459,
342
+ "min_activation_freq": 0.1366167001599475
343
+ },
344
+ "layer_7": {
345
+ "intermediate_size": 11008,
346
+ "dead_count": 0,
347
+ "dead_fraction": 0.0,
348
+ "mean_activation_freq": 0.8904942927683426,
349
+ "min_activation_freq": 0.05687364147151704
350
+ },
351
+ "layer_8": {
352
+ "intermediate_size": 11008,
353
+ "dead_count": 0,
354
+ "dead_fraction": 0.0,
355
+ "mean_activation_freq": 0.9425813152916299,
356
+ "min_activation_freq": 0.06502993889185088
357
+ },
358
+ "layer_9": {
359
+ "intermediate_size": 11008,
360
+ "dead_count": 0,
361
+ "dead_fraction": 0.0,
362
+ "mean_activation_freq": 0.9512113540672161,
363
+ "min_activation_freq": 0.2820766517655744
364
+ },
365
+ "layer_10": {
366
+ "intermediate_size": 11008,
367
+ "dead_count": 0,
368
+ "dead_fraction": 0.0,
369
+ "mean_activation_freq": 0.9742628396775739,
370
+ "min_activation_freq": 0.6589273264159455
371
+ },
372
+ "layer_11": {
373
+ "intermediate_size": 11008,
374
+ "dead_count": 0,
375
+ "dead_fraction": 0.0,
376
+ "mean_activation_freq": 0.9852774333002157,
377
+ "min_activation_freq": 0.9334884960833367
378
+ },
379
+ "layer_12": {
380
+ "intermediate_size": 11008,
381
+ "dead_count": 0,
382
+ "dead_fraction": 0.0,
383
+ "mean_activation_freq": 0.9867415822126716,
384
+ "min_activation_freq": 0.908414674158225
385
+ },
386
+ "layer_13": {
387
+ "intermediate_size": 11008,
388
+ "dead_count": 0,
389
+ "dead_fraction": 0.0,
390
+ "mean_activation_freq": 0.9804252458115417,
391
+ "min_activation_freq": 0.9150996595989009
392
+ },
393
+ "layer_14": {
394
+ "intermediate_size": 11008,
395
+ "dead_count": 0,
396
+ "dead_fraction": 0.0,
397
+ "mean_activation_freq": 0.9827253983881795,
398
+ "min_activation_freq": 0.8576364680310052
399
+ },
400
+ "layer_15": {
401
+ "intermediate_size": 11008,
402
+ "dead_count": 0,
403
+ "dead_fraction": 0.0,
404
+ "mean_activation_freq": 0.9803922596565504,
405
+ "min_activation_freq": 0.9013708321371448
406
+ },
407
+ "layer_16": {
408
+ "intermediate_size": 11008,
409
+ "dead_count": 0,
410
+ "dead_fraction": 0.0,
411
+ "mean_activation_freq": 0.979409905601793,
412
+ "min_activation_freq": 0.9015912726079646
413
+ },
414
+ "layer_17": {
415
+ "intermediate_size": 11008,
416
+ "dead_count": 0,
417
+ "dead_fraction": 0.0,
418
+ "mean_activation_freq": 0.9790893591594924,
419
+ "min_activation_freq": 0.9133105032194562
420
+ },
421
+ "layer_18": {
422
+ "intermediate_size": 11008,
423
+ "dead_count": 0,
424
+ "dead_fraction": 0.0,
425
+ "mean_activation_freq": 0.9798265599054431,
426
+ "min_activation_freq": 0.9193187876799409
427
+ },
428
+ "layer_19": {
429
+ "intermediate_size": 11008,
430
+ "dead_count": 0,
431
+ "dead_fraction": 0.0,
432
+ "mean_activation_freq": 0.976914681576035,
433
+ "min_activation_freq": 0.9159199032112537
434
+ },
435
+ "layer_20": {
436
+ "intermediate_size": 11008,
437
+ "dead_count": 0,
438
+ "dead_fraction": 0.0,
439
+ "mean_activation_freq": 0.9795234603122803,
440
+ "min_activation_freq": 0.9125158922199893
441
+ },
442
+ "layer_21": {
443
+ "intermediate_size": 11008,
444
+ "dead_count": 0,
445
+ "dead_fraction": 0.0,
446
+ "mean_activation_freq": 0.9793927428350289,
447
+ "min_activation_freq": 0.9175603904359595
448
+ },
449
+ "layer_22": {
450
+ "intermediate_size": 11008,
451
+ "dead_count": 0,
452
+ "dead_fraction": 0.0,
453
+ "mean_activation_freq": 0.9787408096464643,
454
+ "min_activation_freq": 0.9210207931755732
455
+ },
456
+ "layer_23": {
457
+ "intermediate_size": 11008,
458
+ "dead_count": 0,
459
+ "dead_fraction": 0.0,
460
+ "mean_activation_freq": 0.978778940952385,
461
+ "min_activation_freq": 0.9328733133740721
462
+ },
463
+ "layer_24": {
464
+ "intermediate_size": 11008,
465
+ "dead_count": 0,
466
+ "dead_fraction": 0.0,
467
+ "mean_activation_freq": 0.9807564346315617,
468
+ "min_activation_freq": 0.9355134725013329
469
+ },
470
+ "layer_25": {
471
+ "intermediate_size": 11008,
472
+ "dead_count": 0,
473
+ "dead_fraction": 0.0,
474
+ "mean_activation_freq": 0.9827950931086782,
475
+ "min_activation_freq": 0.9398402575564943
476
+ },
477
+ "layer_26": {
478
+ "intermediate_size": 11008,
479
+ "dead_count": 0,
480
+ "dead_fraction": 0.0,
481
+ "mean_activation_freq": 0.9850474299646974,
482
+ "min_activation_freq": 0.9503701349300743
483
+ },
484
+ "layer_27": {
485
+ "intermediate_size": 11008,
486
+ "dead_count": 0,
487
+ "dead_fraction": 0.0,
488
+ "mean_activation_freq": 0.9864853771681017,
489
+ "min_activation_freq": 0.9389482426280605
490
+ },
491
+ "layer_28": {
492
+ "intermediate_size": 11008,
493
+ "dead_count": 0,
494
+ "dead_fraction": 0.0,
495
+ "mean_activation_freq": 0.986888830489832,
496
+ "min_activation_freq": 0.2702856498380019
497
+ },
498
+ "layer_29": {
499
+ "intermediate_size": 11008,
500
+ "dead_count": 0,
501
+ "dead_fraction": 0.0,
502
+ "mean_activation_freq": 0.988262459580946,
503
+ "min_activation_freq": 0.9571679038674487
504
+ },
505
+ "layer_30": {
506
+ "intermediate_size": 11008,
507
+ "dead_count": 0,
508
+ "dead_fraction": 0.0,
509
+ "mean_activation_freq": 0.9899264166668257,
510
+ "min_activation_freq": 0.6348890620514293
511
+ },
512
+ "layer_31": {
513
+ "intermediate_size": 11008,
514
+ "dead_count": 0,
515
+ "dead_fraction": 0.0,
516
+ "mean_activation_freq": 0.9907631626708504,
517
+ "min_activation_freq": 0.9352776524627815
518
+ },
519
+ "layer_32": {
520
+ "intermediate_size": 11008,
521
+ "dead_count": 0,
522
+ "dead_fraction": 0.0,
523
+ "mean_activation_freq": 0.9889044736375728,
524
+ "min_activation_freq": 0.9113521715949637
525
+ },
526
+ "layer_33": {
527
+ "intermediate_size": 11008,
528
+ "dead_count": 0,
529
+ "dead_fraction": 0.0,
530
+ "mean_activation_freq": 0.9865240989927934,
531
+ "min_activation_freq": 0.9326016076774802
532
+ },
533
+ "layer_34": {
534
+ "intermediate_size": 11008,
535
+ "dead_count": 0,
536
+ "dead_fraction": 0.0,
537
+ "mean_activation_freq": 0.9876854924708969,
538
+ "min_activation_freq": 0.9066152647336259
539
+ },
540
+ "layer_35": {
541
+ "intermediate_size": 11008,
542
+ "dead_count": 0,
543
+ "dead_fraction": 0.0,
544
+ "mean_activation_freq": 0.9919502344013031,
545
+ "min_activation_freq": 0.007489849485297133
546
+ },
547
+ "_overall": {
548
+ "dead_count": 16778,
549
+ "intermediate_total": 396288,
550
+ "dead_fraction": 0.04233789567183462
551
+ }
552
+ },
553
+ "size": 313,
554
+ "size_pre_filter": 313,
555
+ "skipped_long": 0
556
+ }
557
+ },
558
+ "old_data": {
559
+ "num_records": 52224,
560
+ "source_steps": [
561
+ 25,
562
+ 50,
563
+ 75,
564
+ 100,
565
+ 125,
566
+ 150,
567
+ 175,
568
+ 200,
569
+ 225,
570
+ 250,
571
+ 275,
572
+ 300,
573
+ 325,
574
+ 350,
575
+ 375,
576
+ 400,
577
+ 425,
578
+ 450,
579
+ 475,
580
+ 500,
581
+ 525,
582
+ 550,
583
+ 575,
584
+ 600,
585
+ 625,
586
+ 650,
587
+ 675,
588
+ 700,
589
+ 725,
590
+ 750,
591
+ 775,
592
+ 800,
593
+ 825,
594
+ 850,
595
+ 875,
596
+ 900,
597
+ 925,
598
+ 950,
599
+ 975,
600
+ 1000,
601
+ 1025,
602
+ 1050,
603
+ 1075,
604
+ 1100,
605
+ 1125,
606
+ 1150,
607
+ 1175,
608
+ 1200,
609
+ 1225,
610
+ 1250,
611
+ 1275
612
+ ],
613
+ "num_source_steps": 51,
614
+ "num_positives_pre_cap": 26729,
615
+ "num_negatives_pre_cap": 25495,
616
+ "positive": {
617
+ "loss": 0.5401471858739897,
618
+ "token_entropy": 0.015366265298651233,
619
+ "num_supervised_tokens": 10890999,
620
+ "kl_from_init": 3.6168128850402206,
621
+ "kl_to_init": 0.35471367329119846,
622
+ "dead_units": {
623
+ "layer_0": {
624
+ "intermediate_size": 11008,
625
+ "dead_count": 0,
626
+ "dead_fraction": 0.0,
627
+ "mean_activation_freq": 0.923072584811592,
628
+ "min_activation_freq": 0.777184076502073
629
+ },
630
+ "layer_1": {
631
+ "intermediate_size": 11008,
632
+ "dead_count": 8731,
633
+ "dead_fraction": 0.7931504360465116,
634
+ "mean_activation_freq": 0.1461067420482453,
635
+ "min_activation_freq": 0.0
636
+ },
637
+ "layer_2": {
638
+ "intermediate_size": 11008,
639
+ "dead_count": 3067,
640
+ "dead_fraction": 0.2786155523255814,
641
+ "mean_activation_freq": 0.16644707547897827,
642
+ "min_activation_freq": 0.0
643
+ },
644
+ "layer_3": {
645
+ "intermediate_size": 11008,
646
+ "dead_count": 32,
647
+ "dead_fraction": 0.0029069767441860465,
648
+ "mean_activation_freq": 0.388171961205341,
649
+ "min_activation_freq": 0.0
650
+ },
651
+ "layer_4": {
652
+ "intermediate_size": 11008,
653
+ "dead_count": 0,
654
+ "dead_fraction": 0.0,
655
+ "mean_activation_freq": 0.5587184031797726,
656
+ "min_activation_freq": 9.181894149471504e-08
657
+ },
658
+ "layer_5": {
659
+ "intermediate_size": 11008,
660
+ "dead_count": 0,
661
+ "dead_fraction": 0.0,
662
+ "mean_activation_freq": 0.7205208638381081,
663
+ "min_activation_freq": 2.020016712883731e-06
664
+ },
665
+ "layer_6": {
666
+ "intermediate_size": 11008,
667
+ "dead_count": 0,
668
+ "dead_fraction": 0.0,
669
+ "mean_activation_freq": 0.8738505447457383,
670
+ "min_activation_freq": 0.12884226690315553
671
+ },
672
+ "layer_7": {
673
+ "intermediate_size": 11008,
674
+ "dead_count": 0,
675
+ "dead_fraction": 0.0,
676
+ "mean_activation_freq": 0.8836119567635623,
677
+ "min_activation_freq": 0.04841263873038644
678
+ },
679
+ "layer_8": {
680
+ "intermediate_size": 11008,
681
+ "dead_count": 0,
682
+ "dead_fraction": 0.0,
683
+ "mean_activation_freq": 0.9421992035252132,
684
+ "min_activation_freq": 0.0765674480366769
685
+ },
686
+ "layer_9": {
687
+ "intermediate_size": 11008,
688
+ "dead_count": 0,
689
+ "dead_fraction": 0.0,
690
+ "mean_activation_freq": 0.9503807082281811,
691
+ "min_activation_freq": 0.25814693399567845
692
+ },
693
+ "layer_10": {
694
+ "intermediate_size": 11008,
695
+ "dead_count": 0,
696
+ "dead_fraction": 0.0,
697
+ "mean_activation_freq": 0.9738728335418847,
698
+ "min_activation_freq": 0.6728320331312123
699
+ },
700
+ "layer_11": {
701
+ "intermediate_size": 11008,
702
+ "dead_count": 0,
703
+ "dead_fraction": 0.0,
704
+ "mean_activation_freq": 0.985264744777684,
705
+ "min_activation_freq": 0.9336235362798215
706
+ },
707
+ "layer_12": {
708
+ "intermediate_size": 11008,
709
+ "dead_count": 0,
710
+ "dead_fraction": 0.0,
711
+ "mean_activation_freq": 0.9867527638364195,
712
+ "min_activation_freq": 0.9137106706189212
713
+ },
714
+ "layer_13": {
715
+ "intermediate_size": 11008,
716
+ "dead_count": 0,
717
+ "dead_fraction": 0.0,
718
+ "mean_activation_freq": 0.9807838859613275,
719
+ "min_activation_freq": 0.912307860830765
720
+ },
721
+ "layer_14": {
722
+ "intermediate_size": 11008,
723
+ "dead_count": 0,
724
+ "dead_fraction": 0.0,
725
+ "mean_activation_freq": 0.9828500526441899,
726
+ "min_activation_freq": 0.8728880610493124
727
+ },
728
+ "layer_15": {
729
+ "intermediate_size": 11008,
730
+ "dead_count": 0,
731
+ "dead_fraction": 0.0,
732
+ "mean_activation_freq": 0.9805985510127312,
733
+ "min_activation_freq": 0.905470930628127
734
+ },
735
+ "layer_16": {
736
+ "intermediate_size": 11008,
737
+ "dead_count": 0,
738
+ "dead_fraction": 0.0,
739
+ "mean_activation_freq": 0.9796024806514962,
740
+ "min_activation_freq": 0.909673299942457
741
+ },
742
+ "layer_17": {
743
+ "intermediate_size": 11008,
744
+ "dead_count": 0,
745
+ "dead_fraction": 0.0,
746
+ "mean_activation_freq": 0.9791683927343137,
747
+ "min_activation_freq": 0.914321082941978
748
+ },
749
+ "layer_18": {
750
+ "intermediate_size": 11008,
751
+ "dead_count": 0,
752
+ "dead_fraction": 0.0,
753
+ "mean_activation_freq": 0.9798405845413247,
754
+ "min_activation_freq": 0.9218387587768578
755
+ },
756
+ "layer_19": {
757
+ "intermediate_size": 11008,
758
+ "dead_count": 0,
759
+ "dead_fraction": 0.0,
760
+ "mean_activation_freq": 0.9768802255924584,
761
+ "min_activation_freq": 0.9211772951223299
762
+ },
763
+ "layer_20": {
764
+ "intermediate_size": 11008,
765
+ "dead_count": 0,
766
+ "dead_fraction": 0.0,
767
+ "mean_activation_freq": 0.9794030384248609,
768
+ "min_activation_freq": 0.9086790844439523
769
+ },
770
+ "layer_21": {
771
+ "intermediate_size": 11008,
772
+ "dead_count": 0,
773
+ "dead_fraction": 0.0,
774
+ "mean_activation_freq": 0.9795654684511937,
775
+ "min_activation_freq": 0.9179853932591492
776
+ },
777
+ "layer_22": {
778
+ "intermediate_size": 11008,
779
+ "dead_count": 0,
780
+ "dead_fraction": 0.0,
781
+ "mean_activation_freq": 0.9792202015053231,
782
+ "min_activation_freq": 0.930524096090726
783
+ },
784
+ "layer_23": {
785
+ "intermediate_size": 11008,
786
+ "dead_count": 0,
787
+ "dead_fraction": 0.0,
788
+ "mean_activation_freq": 0.9794770648273323,
789
+ "min_activation_freq": 0.9363634134940239
790
+ },
791
+ "layer_24": {
792
+ "intermediate_size": 11008,
793
+ "dead_count": 0,
794
+ "dead_fraction": 0.0,
795
+ "mean_activation_freq": 0.9812108742257558,
796
+ "min_activation_freq": 0.9331851926531257
797
+ },
798
+ "layer_25": {
799
+ "intermediate_size": 11008,
800
+ "dead_count": 0,
801
+ "dead_fraction": 0.0,
802
+ "mean_activation_freq": 0.9832673341250838,
803
+ "min_activation_freq": 0.9448892613065156
804
+ },
805
+ "layer_26": {
806
+ "intermediate_size": 11008,
807
+ "dead_count": 0,
808
+ "dead_fraction": 0.0,
809
+ "mean_activation_freq": 0.9853870590267348,
810
+ "min_activation_freq": 0.9531230330661127
811
+ },
812
+ "layer_27": {
813
+ "intermediate_size": 11008,
814
+ "dead_count": 0,
815
+ "dead_fraction": 0.0,
816
+ "mean_activation_freq": 0.9867413812154038,
817
+ "min_activation_freq": 0.9375756071596371
818
+ },
819
+ "layer_28": {
820
+ "intermediate_size": 11008,
821
+ "dead_count": 0,
822
+ "dead_fraction": 0.0,
823
+ "mean_activation_freq": 0.9870411182610894,
824
+ "min_activation_freq": 0.26841862716175074
825
+ },
826
+ "layer_29": {
827
+ "intermediate_size": 11008,
828
+ "dead_count": 0,
829
+ "dead_fraction": 0.0,
830
+ "mean_activation_freq": 0.988409949844381,
831
+ "min_activation_freq": 0.9564804844808085
832
+ },
833
+ "layer_30": {
834
+ "intermediate_size": 11008,
835
+ "dead_count": 0,
836
+ "dead_fraction": 0.0,
837
+ "mean_activation_freq": 0.9900316512930156,
838
+ "min_activation_freq": 0.6042004043889821
839
+ },
840
+ "layer_31": {
841
+ "intermediate_size": 11008,
842
+ "dead_count": 0,
843
+ "dead_fraction": 0.0,
844
+ "mean_activation_freq": 0.9908636350500601,
845
+ "min_activation_freq": 0.9425996641814034
846
+ },
847
+ "layer_32": {
848
+ "intermediate_size": 11008,
849
+ "dead_count": 0,
850
+ "dead_fraction": 0.0,
851
+ "mean_activation_freq": 0.9890728715082963,
852
+ "min_activation_freq": 0.9131752743710655
853
+ },
854
+ "layer_33": {
855
+ "intermediate_size": 11008,
856
+ "dead_count": 0,
857
+ "dead_fraction": 0.0,
858
+ "mean_activation_freq": 0.9867600527906706,
859
+ "min_activation_freq": 0.93823615262475
860
+ },
861
+ "layer_34": {
862
+ "intermediate_size": 11008,
863
+ "dead_count": 0,
864
+ "dead_fraction": 0.0,
865
+ "mean_activation_freq": 0.9879444755555177,
866
+ "min_activation_freq": 0.8888811760978034
867
+ },
868
+ "layer_35": {
869
+ "intermediate_size": 11008,
870
+ "dead_count": 0,
871
+ "dead_fraction": 0.0,
872
+ "mean_activation_freq": 0.9915063397395094,
873
+ "min_activation_freq": 0.00865641434729725
874
+ },
875
+ "_overall": {
876
+ "dead_count": 11830,
877
+ "intermediate_total": 396288,
878
+ "dead_fraction": 0.02985202680878553
879
+ }
880
+ },
881
+ "size": 26729,
882
+ "size_pre_filter": 26729,
883
+ "skipped_long": 0
884
+ },
885
+ "negative": {
886
+ "loss": 1.0501002772143966,
887
+ "token_entropy": 0.031723241129886504,
888
+ "num_supervised_tokens": 11713694,
889
+ "kl_from_init": 3.3462305752156736,
890
+ "kl_to_init": 0.3788395098669021,
891
+ "dead_units": {
892
+ "layer_0": {
893
+ "intermediate_size": 11008,
894
+ "dead_count": 0,
895
+ "dead_fraction": 0.0,
896
+ "mean_activation_freq": 0.9233814239938657,
897
+ "min_activation_freq": 0.7794210775866264
898
+ },
899
+ "layer_1": {
900
+ "intermediate_size": 11008,
901
+ "dead_count": 4976,
902
+ "dead_fraction": 0.45203488372093026,
903
+ "mean_activation_freq": 0.14620413655754808,
904
+ "min_activation_freq": 0.0
905
+ },
906
+ "layer_2": {
907
+ "intermediate_size": 11008,
908
+ "dead_count": 842,
909
+ "dead_fraction": 0.07648982558139535,
910
+ "mean_activation_freq": 0.16647381344295545,
911
+ "min_activation_freq": 0.0
912
+ },
913
+ "layer_3": {
914
+ "intermediate_size": 11008,
915
+ "dead_count": 0,
916
+ "dead_fraction": 0.0,
917
+ "mean_activation_freq": 0.38821020389167415,
918
+ "min_activation_freq": 9.390718248231514e-07
919
+ },
920
+ "layer_4": {
921
+ "intermediate_size": 11008,
922
+ "dead_count": 0,
923
+ "dead_fraction": 0.0,
924
+ "mean_activation_freq": 0.5597537910989697,
925
+ "min_activation_freq": 3.2440663039345232e-06
926
+ },
927
+ "layer_5": {
928
+ "intermediate_size": 11008,
929
+ "dead_count": 0,
930
+ "dead_fraction": 0.0,
931
+ "mean_activation_freq": 0.7205096112148068,
932
+ "min_activation_freq": 3.056251938969893e-05
933
+ },
934
+ "layer_6": {
935
+ "intermediate_size": 11008,
936
+ "dead_count": 0,
937
+ "dead_fraction": 0.0,
938
+ "mean_activation_freq": 0.8730098090313166,
939
+ "min_activation_freq": 0.1258415150677489
940
+ },
941
+ "layer_7": {
942
+ "intermediate_size": 11008,
943
+ "dead_count": 0,
944
+ "dead_fraction": 0.0,
945
+ "mean_activation_freq": 0.8848983354743691,
946
+ "min_activation_freq": 0.05157339776845801
947
+ },
948
+ "layer_8": {
949
+ "intermediate_size": 11008,
950
+ "dead_count": 0,
951
+ "dead_fraction": 0.0,
952
+ "mean_activation_freq": 0.9422977845166238,
953
+ "min_activation_freq": 0.07714193319374742
954
+ },
955
+ "layer_9": {
956
+ "intermediate_size": 11008,
957
+ "dead_count": 0,
958
+ "dead_fraction": 0.0,
959
+ "mean_activation_freq": 0.9508083900365739,
960
+ "min_activation_freq": 0.2668677361727223
961
+ },
962
+ "layer_10": {
963
+ "intermediate_size": 11008,
964
+ "dead_count": 0,
965
+ "dead_fraction": 0.0,
966
+ "mean_activation_freq": 0.9739647672125598,
967
+ "min_activation_freq": 0.682935118503181
968
+ },
969
+ "layer_11": {
970
+ "intermediate_size": 11008,
971
+ "dead_count": 0,
972
+ "dead_fraction": 0.0,
973
+ "mean_activation_freq": 0.9852452476038024,
974
+ "min_activation_freq": 0.9328887198180181
975
+ },
976
+ "layer_12": {
977
+ "intermediate_size": 11008,
978
+ "dead_count": 0,
979
+ "dead_fraction": 0.0,
980
+ "mean_activation_freq": 0.9867464854298877,
981
+ "min_activation_freq": 0.9132114941708397
982
+ },
983
+ "layer_13": {
984
+ "intermediate_size": 11008,
985
+ "dead_count": 0,
986
+ "dead_fraction": 0.0,
987
+ "mean_activation_freq": 0.980751892850584,
988
+ "min_activation_freq": 0.9117216140356749
989
+ },
990
+ "layer_14": {
991
+ "intermediate_size": 11008,
992
+ "dead_count": 0,
993
+ "dead_fraction": 0.0,
994
+ "mean_activation_freq": 0.9827984211102249,
995
+ "min_activation_freq": 0.8746616566900245
996
+ },
997
+ "layer_15": {
998
+ "intermediate_size": 11008,
999
+ "dead_count": 0,
1000
+ "dead_fraction": 0.0,
1001
+ "mean_activation_freq": 0.9805842790770393,
1002
+ "min_activation_freq": 0.908410019930519
1003
+ },
1004
+ "layer_16": {
1005
+ "intermediate_size": 11008,
1006
+ "dead_count": 0,
1007
+ "dead_fraction": 0.0,
1008
+ "mean_activation_freq": 0.9796298850756431,
1009
+ "min_activation_freq": 0.906264240810798
1010
+ },
1011
+ "layer_17": {
1012
+ "intermediate_size": 11008,
1013
+ "dead_count": 0,
1014
+ "dead_fraction": 0.0,
1015
+ "mean_activation_freq": 0.9791760215218966,
1016
+ "min_activation_freq": 0.9171079592825286
1017
+ },
1018
+ "layer_18": {
1019
+ "intermediate_size": 11008,
1020
+ "dead_count": 0,
1021
+ "dead_fraction": 0.0,
1022
+ "mean_activation_freq": 0.9798492796936674,
1023
+ "min_activation_freq": 0.9229659746959413
1024
+ },
1025
+ "layer_19": {
1026
+ "intermediate_size": 11008,
1027
+ "dead_count": 0,
1028
+ "dead_fraction": 0.0,
1029
+ "mean_activation_freq": 0.9768640711347145,
1030
+ "min_activation_freq": 0.9227965149166437
1031
+ },
1032
+ "layer_20": {
1033
+ "intermediate_size": 11008,
1034
+ "dead_count": 0,
1035
+ "dead_fraction": 0.0,
1036
+ "mean_activation_freq": 0.9793921972938008,
1037
+ "min_activation_freq": 0.9128887095735982
1038
+ },
1039
+ "layer_21": {
1040
+ "intermediate_size": 11008,
1041
+ "dead_count": 0,
1042
+ "dead_fraction": 0.0,
1043
+ "mean_activation_freq": 0.979425614176475,
1044
+ "min_activation_freq": 0.9163229806071423
1045
+ },
1046
+ "layer_22": {
1047
+ "intermediate_size": 11008,
1048
+ "dead_count": 0,
1049
+ "dead_fraction": 0.0,
1050
+ "mean_activation_freq": 0.9789843089014045,
1051
+ "min_activation_freq": 0.9307004263556825
1052
+ },
1053
+ "layer_23": {
1054
+ "intermediate_size": 11008,
1055
+ "dead_count": 0,
1056
+ "dead_fraction": 0.0,
1057
+ "mean_activation_freq": 0.9791408052659919,
1058
+ "min_activation_freq": 0.9357478520439411
1059
+ },
1060
+ "layer_24": {
1061
+ "intermediate_size": 11008,
1062
+ "dead_count": 0,
1063
+ "dead_fraction": 0.0,
1064
+ "mean_activation_freq": 0.9809966780201427,
1065
+ "min_activation_freq": 0.9322701275959573
1066
+ },
1067
+ "layer_25": {
1068
+ "intermediate_size": 11008,
1069
+ "dead_count": 0,
1070
+ "dead_fraction": 0.0,
1071
+ "mean_activation_freq": 0.9830069956413685,
1072
+ "min_activation_freq": 0.9434786327865488
1073
+ },
1074
+ "layer_26": {
1075
+ "intermediate_size": 11008,
1076
+ "dead_count": 0,
1077
+ "dead_fraction": 0.0,
1078
+ "mean_activation_freq": 0.9851738557788767,
1079
+ "min_activation_freq": 0.9522614300834562
1080
+ },
1081
+ "layer_27": {
1082
+ "intermediate_size": 11008,
1083
+ "dead_count": 0,
1084
+ "dead_fraction": 0.0,
1085
+ "mean_activation_freq": 0.9865963983654465,
1086
+ "min_activation_freq": 0.9358695899005045
1087
+ },
1088
+ "layer_28": {
1089
+ "intermediate_size": 11008,
1090
+ "dead_count": 0,
1091
+ "dead_fraction": 0.0,
1092
+ "mean_activation_freq": 0.9869381775841854,
1093
+ "min_activation_freq": 0.2858502194098634
1094
+ },
1095
+ "layer_29": {
1096
+ "intermediate_size": 11008,
1097
+ "dead_count": 0,
1098
+ "dead_fraction": 0.0,
1099
+ "mean_activation_freq": 0.9883171966563039,
1100
+ "min_activation_freq": 0.9568379539366488
1101
+ },
1102
+ "layer_30": {
1103
+ "intermediate_size": 11008,
1104
+ "dead_count": 0,
1105
+ "dead_fraction": 0.0,
1106
+ "mean_activation_freq": 0.9899665228783125,
1107
+ "min_activation_freq": 0.6186793850001545
1108
+ },
1109
+ "layer_31": {
1110
+ "intermediate_size": 11008,
1111
+ "dead_count": 0,
1112
+ "dead_fraction": 0.0,
1113
+ "mean_activation_freq": 0.9908500578195257,
1114
+ "min_activation_freq": 0.9437283405217859
1115
+ },
1116
+ "layer_32": {
1117
+ "intermediate_size": 11008,
1118
+ "dead_count": 0,
1119
+ "dead_fraction": 0.0,
1120
+ "mean_activation_freq": 0.989079460864973,
1121
+ "min_activation_freq": 0.9093610435785671
1122
+ },
1123
+ "layer_33": {
1124
+ "intermediate_size": 11008,
1125
+ "dead_count": 0,
1126
+ "dead_fraction": 0.0,
1127
+ "mean_activation_freq": 0.9868177727058302,
1128
+ "min_activation_freq": 0.9354037249052263
1129
+ },
1130
+ "layer_34": {
1131
+ "intermediate_size": 11008,
1132
+ "dead_count": 0,
1133
+ "dead_fraction": 0.0,
1134
+ "mean_activation_freq": 0.9879361257256494,
1135
+ "min_activation_freq": 0.9042676887410581
1136
+ },
1137
+ "layer_35": {
1138
+ "intermediate_size": 11008,
1139
+ "dead_count": 0,
1140
+ "dead_fraction": 0.0,
1141
+ "mean_activation_freq": 0.9918638372641997,
1142
+ "min_activation_freq": 0.01670403887962243
1143
+ },
1144
+ "_overall": {
1145
+ "dead_count": 5818,
1146
+ "intermediate_total": 396288,
1147
+ "dead_fraction": 0.014681241925064599
1148
+ }
1149
+ },
1150
+ "size": 25495,
1151
+ "size_pre_filter": 25495,
1152
+ "skipped_long": 0
1153
+ }
1154
+ }
1155
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "eos_token_id": 151643,
4
+ "max_new_tokens": 2048,
5
+ "transformers_version": "4.57.6"
6
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fbb88dc0dc030d485b2677760529051f344e685f225d2a8e0d24de5d31ecaa02
3
+ size 4978979048
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:40978f480f12dd8b3433ba5a0a27f20358162d20347abe5033f023aa2335f1ff
3
+ size 1815277960
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/model.safetensors.index.json ADDED
@@ -0,0 +1,443 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_parameters": 3397103616,
4
+ "total_size": 6794207232
5
+ },
6
+ "weight_map": {
7
+ "lm_head.weight": "model-00001-of-00002.safetensors",
8
+ "model.embed_tokens.weight": "model-00001-of-00002.safetensors",
9
+ "model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
10
+ "model.layers.0.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
11
+ "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
12
+ "model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
13
+ "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
14
+ "model.layers.0.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
15
+ "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
16
+ "model.layers.0.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
17
+ "model.layers.0.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
18
+ "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
19
+ "model.layers.0.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
20
+ "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
21
+ "model.layers.1.input_layernorm.weight": "model-00002-of-00002.safetensors",
22
+ "model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
23
+ "model.layers.1.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
24
+ "model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
25
+ "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
26
+ "model.layers.1.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
27
+ "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
28
+ "model.layers.1.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
29
+ "model.layers.1.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
30
+ "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
31
+ "model.layers.1.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
32
+ "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
33
+ "model.layers.10.input_layernorm.weight": "model-00002-of-00002.safetensors",
34
+ "model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
35
+ "model.layers.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
36
+ "model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
37
+ "model.layers.10.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
38
+ "model.layers.10.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
39
+ "model.layers.10.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
40
+ "model.layers.10.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
41
+ "model.layers.10.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
42
+ "model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
43
+ "model.layers.10.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
44
+ "model.layers.10.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
45
+ "model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
46
+ "model.layers.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
47
+ "model.layers.11.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
48
+ "model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
49
+ "model.layers.11.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
50
+ "model.layers.11.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
51
+ "model.layers.11.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
52
+ "model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
53
+ "model.layers.11.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
54
+ "model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
55
+ "model.layers.11.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
56
+ "model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
57
+ "model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
58
+ "model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
59
+ "model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
60
+ "model.layers.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
61
+ "model.layers.12.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
62
+ "model.layers.12.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
63
+ "model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
64
+ "model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
65
+ "model.layers.12.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
66
+ "model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
67
+ "model.layers.12.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
68
+ "model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
69
+ "model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
70
+ "model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
71
+ "model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
72
+ "model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
73
+ "model.layers.13.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
74
+ "model.layers.13.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
75
+ "model.layers.13.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
76
+ "model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
77
+ "model.layers.13.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
78
+ "model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
79
+ "model.layers.13.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
80
+ "model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
81
+ "model.layers.14.input_layernorm.weight": "model-00001-of-00002.safetensors",
82
+ "model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
83
+ "model.layers.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
84
+ "model.layers.14.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
85
+ "model.layers.14.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
86
+ "model.layers.14.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
87
+ "model.layers.14.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
88
+ "model.layers.14.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
89
+ "model.layers.14.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
90
+ "model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
91
+ "model.layers.14.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
92
+ "model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
93
+ "model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
94
+ "model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
95
+ "model.layers.15.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
96
+ "model.layers.15.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
97
+ "model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
98
+ "model.layers.15.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
99
+ "model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
100
+ "model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
101
+ "model.layers.15.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
102
+ "model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
103
+ "model.layers.15.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
104
+ "model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
105
+ "model.layers.16.input_layernorm.weight": "model-00002-of-00002.safetensors",
106
+ "model.layers.16.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
107
+ "model.layers.16.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
108
+ "model.layers.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
109
+ "model.layers.16.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
110
+ "model.layers.16.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
111
+ "model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
112
+ "model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
113
+ "model.layers.16.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
114
+ "model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
115
+ "model.layers.16.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
116
+ "model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
117
+ "model.layers.17.input_layernorm.weight": "model-00001-of-00002.safetensors",
118
+ "model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
119
+ "model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
120
+ "model.layers.17.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
121
+ "model.layers.17.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
122
+ "model.layers.17.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
123
+ "model.layers.17.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
124
+ "model.layers.17.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
125
+ "model.layers.17.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
126
+ "model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
127
+ "model.layers.17.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
128
+ "model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
129
+ "model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
130
+ "model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
131
+ "model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
132
+ "model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
133
+ "model.layers.18.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
134
+ "model.layers.18.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
135
+ "model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
136
+ "model.layers.18.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
137
+ "model.layers.18.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
138
+ "model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
139
+ "model.layers.18.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
140
+ "model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
141
+ "model.layers.19.input_layernorm.weight": "model-00001-of-00002.safetensors",
142
+ "model.layers.19.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
143
+ "model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
144
+ "model.layers.19.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
145
+ "model.layers.19.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
146
+ "model.layers.19.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
147
+ "model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
148
+ "model.layers.19.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
149
+ "model.layers.19.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
150
+ "model.layers.19.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
151
+ "model.layers.19.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
152
+ "model.layers.19.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
153
+ "model.layers.2.input_layernorm.weight": "model-00001-of-00002.safetensors",
154
+ "model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
155
+ "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
156
+ "model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
157
+ "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
158
+ "model.layers.2.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
159
+ "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
160
+ "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
161
+ "model.layers.2.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
162
+ "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
163
+ "model.layers.2.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
164
+ "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
165
+ "model.layers.20.input_layernorm.weight": "model-00002-of-00002.safetensors",
166
+ "model.layers.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
167
+ "model.layers.20.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
168
+ "model.layers.20.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
169
+ "model.layers.20.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
170
+ "model.layers.20.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
171
+ "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
172
+ "model.layers.20.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
173
+ "model.layers.20.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
174
+ "model.layers.20.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
175
+ "model.layers.20.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
176
+ "model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
177
+ "model.layers.21.input_layernorm.weight": "model-00001-of-00002.safetensors",
178
+ "model.layers.21.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
179
+ "model.layers.21.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
180
+ "model.layers.21.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
181
+ "model.layers.21.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
182
+ "model.layers.21.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
183
+ "model.layers.21.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
184
+ "model.layers.21.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
185
+ "model.layers.21.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
186
+ "model.layers.21.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
187
+ "model.layers.21.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
188
+ "model.layers.21.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
189
+ "model.layers.22.input_layernorm.weight": "model-00002-of-00002.safetensors",
190
+ "model.layers.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
191
+ "model.layers.22.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
192
+ "model.layers.22.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
193
+ "model.layers.22.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
194
+ "model.layers.22.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
195
+ "model.layers.22.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
196
+ "model.layers.22.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
197
+ "model.layers.22.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
198
+ "model.layers.22.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
199
+ "model.layers.22.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
200
+ "model.layers.22.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
201
+ "model.layers.23.input_layernorm.weight": "model-00002-of-00002.safetensors",
202
+ "model.layers.23.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
203
+ "model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
204
+ "model.layers.23.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
205
+ "model.layers.23.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
206
+ "model.layers.23.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
207
+ "model.layers.23.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
208
+ "model.layers.23.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
209
+ "model.layers.23.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
210
+ "model.layers.23.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
211
+ "model.layers.23.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
212
+ "model.layers.23.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
213
+ "model.layers.24.input_layernorm.weight": "model-00001-of-00002.safetensors",
214
+ "model.layers.24.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
215
+ "model.layers.24.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
216
+ "model.layers.24.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
217
+ "model.layers.24.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
218
+ "model.layers.24.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
219
+ "model.layers.24.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
220
+ "model.layers.24.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
221
+ "model.layers.24.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
222
+ "model.layers.24.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
223
+ "model.layers.24.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
224
+ "model.layers.24.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
225
+ "model.layers.25.input_layernorm.weight": "model-00001-of-00002.safetensors",
226
+ "model.layers.25.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
227
+ "model.layers.25.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
228
+ "model.layers.25.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
229
+ "model.layers.25.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
230
+ "model.layers.25.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
231
+ "model.layers.25.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
232
+ "model.layers.25.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
233
+ "model.layers.25.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
234
+ "model.layers.25.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
235
+ "model.layers.25.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
236
+ "model.layers.25.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
237
+ "model.layers.26.input_layernorm.weight": "model-00001-of-00002.safetensors",
238
+ "model.layers.26.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
239
+ "model.layers.26.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
240
+ "model.layers.26.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
241
+ "model.layers.26.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
242
+ "model.layers.26.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
243
+ "model.layers.26.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
244
+ "model.layers.26.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
245
+ "model.layers.26.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
246
+ "model.layers.26.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
247
+ "model.layers.26.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
248
+ "model.layers.26.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
249
+ "model.layers.27.input_layernorm.weight": "model-00002-of-00002.safetensors",
250
+ "model.layers.27.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
251
+ "model.layers.27.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
252
+ "model.layers.27.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
253
+ "model.layers.27.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
254
+ "model.layers.27.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
255
+ "model.layers.27.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
256
+ "model.layers.27.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
257
+ "model.layers.27.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
258
+ "model.layers.27.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
259
+ "model.layers.27.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
260
+ "model.layers.27.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
261
+ "model.layers.28.input_layernorm.weight": "model-00001-of-00002.safetensors",
262
+ "model.layers.28.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
263
+ "model.layers.28.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
264
+ "model.layers.28.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
265
+ "model.layers.28.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
266
+ "model.layers.28.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
267
+ "model.layers.28.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
268
+ "model.layers.28.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
269
+ "model.layers.28.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
270
+ "model.layers.28.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
271
+ "model.layers.28.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
272
+ "model.layers.28.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
273
+ "model.layers.29.input_layernorm.weight": "model-00002-of-00002.safetensors",
274
+ "model.layers.29.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
275
+ "model.layers.29.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
276
+ "model.layers.29.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
277
+ "model.layers.29.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
278
+ "model.layers.29.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
279
+ "model.layers.29.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
280
+ "model.layers.29.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
281
+ "model.layers.29.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
282
+ "model.layers.29.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
283
+ "model.layers.29.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
284
+ "model.layers.29.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
285
+ "model.layers.3.input_layernorm.weight": "model-00002-of-00002.safetensors",
286
+ "model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
287
+ "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
288
+ "model.layers.3.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
289
+ "model.layers.3.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
290
+ "model.layers.3.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
291
+ "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
292
+ "model.layers.3.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
293
+ "model.layers.3.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
294
+ "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
295
+ "model.layers.3.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
296
+ "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
297
+ "model.layers.30.input_layernorm.weight": "model-00001-of-00002.safetensors",
298
+ "model.layers.30.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
299
+ "model.layers.30.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
300
+ "model.layers.30.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
301
+ "model.layers.30.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
302
+ "model.layers.30.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
303
+ "model.layers.30.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
304
+ "model.layers.30.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
305
+ "model.layers.30.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
306
+ "model.layers.30.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
307
+ "model.layers.30.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
308
+ "model.layers.30.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
309
+ "model.layers.31.input_layernorm.weight": "model-00002-of-00002.safetensors",
310
+ "model.layers.31.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
311
+ "model.layers.31.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
312
+ "model.layers.31.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
313
+ "model.layers.31.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
314
+ "model.layers.31.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
315
+ "model.layers.31.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
316
+ "model.layers.31.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
317
+ "model.layers.31.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
318
+ "model.layers.31.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
319
+ "model.layers.31.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
320
+ "model.layers.31.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
321
+ "model.layers.32.input_layernorm.weight": "model-00002-of-00002.safetensors",
322
+ "model.layers.32.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
323
+ "model.layers.32.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
324
+ "model.layers.32.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
325
+ "model.layers.32.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
326
+ "model.layers.32.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
327
+ "model.layers.32.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
328
+ "model.layers.32.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
329
+ "model.layers.32.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
330
+ "model.layers.32.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
331
+ "model.layers.32.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
332
+ "model.layers.32.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
333
+ "model.layers.33.input_layernorm.weight": "model-00002-of-00002.safetensors",
334
+ "model.layers.33.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
335
+ "model.layers.33.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
336
+ "model.layers.33.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
337
+ "model.layers.33.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
338
+ "model.layers.33.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
339
+ "model.layers.33.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
340
+ "model.layers.33.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
341
+ "model.layers.33.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
342
+ "model.layers.33.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
343
+ "model.layers.33.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
344
+ "model.layers.33.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
345
+ "model.layers.34.input_layernorm.weight": "model-00002-of-00002.safetensors",
346
+ "model.layers.34.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
347
+ "model.layers.34.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
348
+ "model.layers.34.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
349
+ "model.layers.34.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
350
+ "model.layers.34.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
351
+ "model.layers.34.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
352
+ "model.layers.34.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
353
+ "model.layers.34.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
354
+ "model.layers.34.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
355
+ "model.layers.34.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
356
+ "model.layers.34.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
357
+ "model.layers.35.input_layernorm.weight": "model-00001-of-00002.safetensors",
358
+ "model.layers.35.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
359
+ "model.layers.35.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
360
+ "model.layers.35.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
361
+ "model.layers.35.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
362
+ "model.layers.35.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
363
+ "model.layers.35.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
364
+ "model.layers.35.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
365
+ "model.layers.35.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
366
+ "model.layers.35.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
367
+ "model.layers.35.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
368
+ "model.layers.35.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
369
+ "model.layers.4.input_layernorm.weight": "model-00002-of-00002.safetensors",
370
+ "model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
371
+ "model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
372
+ "model.layers.4.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
373
+ "model.layers.4.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
374
+ "model.layers.4.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
375
+ "model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
376
+ "model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
377
+ "model.layers.4.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
378
+ "model.layers.4.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
379
+ "model.layers.4.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
380
+ "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
381
+ "model.layers.5.input_layernorm.weight": "model-00002-of-00002.safetensors",
382
+ "model.layers.5.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
383
+ "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
384
+ "model.layers.5.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
385
+ "model.layers.5.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
386
+ "model.layers.5.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
387
+ "model.layers.5.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
388
+ "model.layers.5.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
389
+ "model.layers.5.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
390
+ "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
391
+ "model.layers.5.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
392
+ "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
393
+ "model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
394
+ "model.layers.6.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
395
+ "model.layers.6.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
396
+ "model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
397
+ "model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
398
+ "model.layers.6.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
399
+ "model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
400
+ "model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
401
+ "model.layers.6.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
402
+ "model.layers.6.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
403
+ "model.layers.6.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
404
+ "model.layers.6.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
405
+ "model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
406
+ "model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
407
+ "model.layers.7.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
408
+ "model.layers.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
409
+ "model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
410
+ "model.layers.7.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
411
+ "model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
412
+ "model.layers.7.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
413
+ "model.layers.7.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
414
+ "model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
415
+ "model.layers.7.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
416
+ "model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
417
+ "model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
418
+ "model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
419
+ "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
420
+ "model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
421
+ "model.layers.8.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
422
+ "model.layers.8.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
423
+ "model.layers.8.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
424
+ "model.layers.8.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
425
+ "model.layers.8.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
426
+ "model.layers.8.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
427
+ "model.layers.8.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
428
+ "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
429
+ "model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
430
+ "model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
431
+ "model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
432
+ "model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
433
+ "model.layers.9.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
434
+ "model.layers.9.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
435
+ "model.layers.9.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
436
+ "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
437
+ "model.layers.9.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
438
+ "model.layers.9.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
439
+ "model.layers.9.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
440
+ "model.layers.9.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
441
+ "model.norm.weight": "model-00001-of-00002.safetensors"
442
+ }
443
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|endoftext|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/tokenizer_config.json ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_bos_token": false,
3
+ "add_prefix_space": false,
4
+ "added_tokens_decoder": {
5
+ "151643": {
6
+ "content": "<|endoftext|>",
7
+ "lstrip": false,
8
+ "normalized": false,
9
+ "rstrip": false,
10
+ "single_word": false,
11
+ "special": true
12
+ },
13
+ "151644": {
14
+ "content": "<|im_start|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false,
19
+ "special": true
20
+ },
21
+ "151645": {
22
+ "content": "<|im_end|>",
23
+ "lstrip": false,
24
+ "normalized": false,
25
+ "rstrip": false,
26
+ "single_word": false,
27
+ "special": true
28
+ },
29
+ "151646": {
30
+ "content": "<|object_ref_start|>",
31
+ "lstrip": false,
32
+ "normalized": false,
33
+ "rstrip": false,
34
+ "single_word": false,
35
+ "special": true
36
+ },
37
+ "151647": {
38
+ "content": "<|object_ref_end|>",
39
+ "lstrip": false,
40
+ "normalized": false,
41
+ "rstrip": false,
42
+ "single_word": false,
43
+ "special": true
44
+ },
45
+ "151648": {
46
+ "content": "<|box_start|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": false,
50
+ "single_word": false,
51
+ "special": true
52
+ },
53
+ "151649": {
54
+ "content": "<|box_end|>",
55
+ "lstrip": false,
56
+ "normalized": false,
57
+ "rstrip": false,
58
+ "single_word": false,
59
+ "special": true
60
+ },
61
+ "151650": {
62
+ "content": "<|quad_start|>",
63
+ "lstrip": false,
64
+ "normalized": false,
65
+ "rstrip": false,
66
+ "single_word": false,
67
+ "special": true
68
+ },
69
+ "151651": {
70
+ "content": "<|quad_end|>",
71
+ "lstrip": false,
72
+ "normalized": false,
73
+ "rstrip": false,
74
+ "single_word": false,
75
+ "special": true
76
+ },
77
+ "151652": {
78
+ "content": "<|vision_start|>",
79
+ "lstrip": false,
80
+ "normalized": false,
81
+ "rstrip": false,
82
+ "single_word": false,
83
+ "special": true
84
+ },
85
+ "151653": {
86
+ "content": "<|vision_end|>",
87
+ "lstrip": false,
88
+ "normalized": false,
89
+ "rstrip": false,
90
+ "single_word": false,
91
+ "special": true
92
+ },
93
+ "151654": {
94
+ "content": "<|vision_pad|>",
95
+ "lstrip": false,
96
+ "normalized": false,
97
+ "rstrip": false,
98
+ "single_word": false,
99
+ "special": true
100
+ },
101
+ "151655": {
102
+ "content": "<|image_pad|>",
103
+ "lstrip": false,
104
+ "normalized": false,
105
+ "rstrip": false,
106
+ "single_word": false,
107
+ "special": true
108
+ },
109
+ "151656": {
110
+ "content": "<|video_pad|>",
111
+ "lstrip": false,
112
+ "normalized": false,
113
+ "rstrip": false,
114
+ "single_word": false,
115
+ "special": true
116
+ },
117
+ "151657": {
118
+ "content": "<tool_call>",
119
+ "lstrip": false,
120
+ "normalized": false,
121
+ "rstrip": false,
122
+ "single_word": false,
123
+ "special": false
124
+ },
125
+ "151658": {
126
+ "content": "</tool_call>",
127
+ "lstrip": false,
128
+ "normalized": false,
129
+ "rstrip": false,
130
+ "single_word": false,
131
+ "special": false
132
+ },
133
+ "151659": {
134
+ "content": "<|fim_prefix|>",
135
+ "lstrip": false,
136
+ "normalized": false,
137
+ "rstrip": false,
138
+ "single_word": false,
139
+ "special": false
140
+ },
141
+ "151660": {
142
+ "content": "<|fim_middle|>",
143
+ "lstrip": false,
144
+ "normalized": false,
145
+ "rstrip": false,
146
+ "single_word": false,
147
+ "special": false
148
+ },
149
+ "151661": {
150
+ "content": "<|fim_suffix|>",
151
+ "lstrip": false,
152
+ "normalized": false,
153
+ "rstrip": false,
154
+ "single_word": false,
155
+ "special": false
156
+ },
157
+ "151662": {
158
+ "content": "<|fim_pad|>",
159
+ "lstrip": false,
160
+ "normalized": false,
161
+ "rstrip": false,
162
+ "single_word": false,
163
+ "special": false
164
+ },
165
+ "151663": {
166
+ "content": "<|repo_name|>",
167
+ "lstrip": false,
168
+ "normalized": false,
169
+ "rstrip": false,
170
+ "single_word": false,
171
+ "special": false
172
+ },
173
+ "151664": {
174
+ "content": "<|file_sep|>",
175
+ "lstrip": false,
176
+ "normalized": false,
177
+ "rstrip": false,
178
+ "single_word": false,
179
+ "special": false
180
+ }
181
+ },
182
+ "additional_special_tokens": [
183
+ "<|im_start|>",
184
+ "<|im_end|>",
185
+ "<|object_ref_start|>",
186
+ "<|object_ref_end|>",
187
+ "<|box_start|>",
188
+ "<|box_end|>",
189
+ "<|quad_start|>",
190
+ "<|quad_end|>",
191
+ "<|vision_start|>",
192
+ "<|vision_end|>",
193
+ "<|vision_pad|>",
194
+ "<|image_pad|>",
195
+ "<|video_pad|>"
196
+ ],
197
+ "bos_token": null,
198
+ "clean_up_tokenization_spaces": false,
199
+ "eos_token": "<|endoftext|>",
200
+ "errors": "replace",
201
+ "extra_special_tokens": {},
202
+ "model_max_length": 131072,
203
+ "pad_token": "<|endoftext|>",
204
+ "split_special_tokens": false,
205
+ "tokenizer_class": "Qwen2Tokenizer",
206
+ "unk_token": null
207
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1300/vocab.json ADDED
The diff for this file is too large to render. See raw diff
 
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/added_tokens.json ADDED
@@ -0,0 +1,24 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "</tool_call>": 151658,
3
+ "<tool_call>": 151657,
4
+ "<|box_end|>": 151649,
5
+ "<|box_start|>": 151648,
6
+ "<|endoftext|>": 151643,
7
+ "<|file_sep|>": 151664,
8
+ "<|fim_middle|>": 151660,
9
+ "<|fim_pad|>": 151662,
10
+ "<|fim_prefix|>": 151659,
11
+ "<|fim_suffix|>": 151661,
12
+ "<|im_end|>": 151645,
13
+ "<|im_start|>": 151644,
14
+ "<|image_pad|>": 151655,
15
+ "<|object_ref_end|>": 151647,
16
+ "<|object_ref_start|>": 151646,
17
+ "<|quad_end|>": 151651,
18
+ "<|quad_start|>": 151650,
19
+ "<|repo_name|>": 151663,
20
+ "<|video_pad|>": 151656,
21
+ "<|vision_end|>": 151653,
22
+ "<|vision_pad|>": 151654,
23
+ "<|vision_start|>": 151652
24
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/config.json ADDED
@@ -0,0 +1,67 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "dtype": "bfloat16",
7
+ "eos_token_id": 151643,
8
+ "hidden_act": "silu",
9
+ "hidden_size": 2048,
10
+ "initializer_range": 0.02,
11
+ "intermediate_size": 11008,
12
+ "layer_types": [
13
+ "full_attention",
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention",
38
+ "full_attention",
39
+ "full_attention",
40
+ "full_attention",
41
+ "full_attention",
42
+ "full_attention",
43
+ "full_attention",
44
+ "full_attention",
45
+ "full_attention",
46
+ "full_attention",
47
+ "full_attention",
48
+ "full_attention"
49
+ ],
50
+ "max_position_embeddings": 32768,
51
+ "max_window_layers": 36,
52
+ "model_type": "qwen2",
53
+ "num_attention_heads": 16,
54
+ "num_hidden_layers": 36,
55
+ "num_key_value_heads": 2,
56
+ "pad_token_id": 151643,
57
+ "rms_norm_eps": 1e-06,
58
+ "rope_scaling": null,
59
+ "rope_theta": 1000000.0,
60
+ "sliding_window": null,
61
+ "tie_word_embeddings": true,
62
+ "transformers_version": "4.57.6",
63
+ "use_cache": true,
64
+ "use_mrope": false,
65
+ "use_sliding_window": false,
66
+ "vocab_size": 151936
67
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/eval_metrics.json ADDED
@@ -0,0 +1,1159 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "global_step": 1400,
3
+ "ckpt_dir": "/mnt/mystorageoutput/projects/jy_intern/amlt-results/7213678591.49431-3a69baee-2340/qwen3b_kk_rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400",
4
+ "rollouts_dir": "/mnt/mystorageoutput/projects/jy_intern/amlt-results/7213678591.49431-3a69baee-2340/qwen3b_kk_rlrb_bs128_nobuf_seed2/rollouts/training",
5
+ "init_model_path": "Qwen/Qwen2.5-3B",
6
+ "prev_step": null,
7
+ "positive_score": 1.0,
8
+ "negative_score": 0.0,
9
+ "history_step_stride": 25,
10
+ "max_samples_per_class": 0,
11
+ "in_batch": {
12
+ "num_records": 1024,
13
+ "source_steps": [
14
+ 1400
15
+ ],
16
+ "num_source_steps": 1,
17
+ "num_positives_pre_cap": 770,
18
+ "num_negatives_pre_cap": 254,
19
+ "positive": {
20
+ "loss": 0.05449246197002923,
21
+ "token_entropy": 0.013225688812733314,
22
+ "num_supervised_tokens": 378198,
23
+ "kl_from_init": 3.522097292570211,
24
+ "kl_to_init": 0.3223560613999825,
25
+ "dead_units": {
26
+ "layer_0": {
27
+ "intermediate_size": 11008,
28
+ "dead_count": 0,
29
+ "dead_fraction": 0.0,
30
+ "mean_activation_freq": 0.9230237785556249,
31
+ "min_activation_freq": 0.7752685101454793
32
+ },
33
+ "layer_1": {
34
+ "intermediate_size": 11008,
35
+ "dead_count": 8840,
36
+ "dead_fraction": 0.8030523255813954,
37
+ "mean_activation_freq": 0.14606047716161843,
38
+ "min_activation_freq": 0.0
39
+ },
40
+ "layer_2": {
41
+ "intermediate_size": 11008,
42
+ "dead_count": 7301,
43
+ "dead_fraction": 0.6632449127906976,
44
+ "mean_activation_freq": 0.16620906689726753,
45
+ "min_activation_freq": 0.0
46
+ },
47
+ "layer_3": {
48
+ "intermediate_size": 11008,
49
+ "dead_count": 252,
50
+ "dead_fraction": 0.022892441860465115,
51
+ "mean_activation_freq": 0.38756262802064295,
52
+ "min_activation_freq": 0.0
53
+ },
54
+ "layer_4": {
55
+ "intermediate_size": 11008,
56
+ "dead_count": 24,
57
+ "dead_fraction": 0.002180232558139535,
58
+ "mean_activation_freq": 0.5616341499575341,
59
+ "min_activation_freq": 0.0
60
+ },
61
+ "layer_5": {
62
+ "intermediate_size": 11008,
63
+ "dead_count": 22,
64
+ "dead_fraction": 0.001998546511627907,
65
+ "mean_activation_freq": 0.7230543182103486,
66
+ "min_activation_freq": 0.0
67
+ },
68
+ "layer_6": {
69
+ "intermediate_size": 11008,
70
+ "dead_count": 0,
71
+ "dead_fraction": 0.0,
72
+ "mean_activation_freq": 0.8750985774516472,
73
+ "min_activation_freq": 0.1351117668522837
74
+ },
75
+ "layer_7": {
76
+ "intermediate_size": 11008,
77
+ "dead_count": 0,
78
+ "dead_fraction": 0.0,
79
+ "mean_activation_freq": 0.8854423677878924,
80
+ "min_activation_freq": 0.05837154083310858
81
+ },
82
+ "layer_8": {
83
+ "intermediate_size": 11008,
84
+ "dead_count": 0,
85
+ "dead_fraction": 0.0,
86
+ "mean_activation_freq": 0.9423164236015416,
87
+ "min_activation_freq": 0.0712748348748539
88
+ },
89
+ "layer_9": {
90
+ "intermediate_size": 11008,
91
+ "dead_count": 0,
92
+ "dead_fraction": 0.0,
93
+ "mean_activation_freq": 0.9505303603716343,
94
+ "min_activation_freq": 0.2670902543112338
95
+ },
96
+ "layer_10": {
97
+ "intermediate_size": 11008,
98
+ "dead_count": 0,
99
+ "dead_fraction": 0.0,
100
+ "mean_activation_freq": 0.9739920710540971,
101
+ "min_activation_freq": 0.6668068049011364
102
+ },
103
+ "layer_11": {
104
+ "intermediate_size": 11008,
105
+ "dead_count": 0,
106
+ "dead_fraction": 0.0,
107
+ "mean_activation_freq": 0.9852190084490472,
108
+ "min_activation_freq": 0.9337172592134279
109
+ },
110
+ "layer_12": {
111
+ "intermediate_size": 11008,
112
+ "dead_count": 0,
113
+ "dead_fraction": 0.0,
114
+ "mean_activation_freq": 0.9867034061911492,
115
+ "min_activation_freq": 0.9129397828650601
116
+ },
117
+ "layer_13": {
118
+ "intermediate_size": 11008,
119
+ "dead_count": 0,
120
+ "dead_fraction": 0.0,
121
+ "mean_activation_freq": 0.9803464489907586,
122
+ "min_activation_freq": 0.9096293475904156
123
+ },
124
+ "layer_14": {
125
+ "intermediate_size": 11008,
126
+ "dead_count": 0,
127
+ "dead_fraction": 0.0,
128
+ "mean_activation_freq": 0.9827137081941943,
129
+ "min_activation_freq": 0.8723790183977704
130
+ },
131
+ "layer_15": {
132
+ "intermediate_size": 11008,
133
+ "dead_count": 0,
134
+ "dead_fraction": 0.0,
135
+ "mean_activation_freq": 0.9804633082771672,
136
+ "min_activation_freq": 0.9046610505608174
137
+ },
138
+ "layer_16": {
139
+ "intermediate_size": 11008,
140
+ "dead_count": 0,
141
+ "dead_fraction": 0.0,
142
+ "mean_activation_freq": 0.9794095176297772,
143
+ "min_activation_freq": 0.9082887799512425
144
+ },
145
+ "layer_17": {
146
+ "intermediate_size": 11008,
147
+ "dead_count": 0,
148
+ "dead_fraction": 0.0,
149
+ "mean_activation_freq": 0.9789243969866835,
150
+ "min_activation_freq": 0.9133073152158393
151
+ },
152
+ "layer_18": {
153
+ "intermediate_size": 11008,
154
+ "dead_count": 0,
155
+ "dead_fraction": 0.0,
156
+ "mean_activation_freq": 0.979623709893501,
157
+ "min_activation_freq": 0.9193359034156711
158
+ },
159
+ "layer_19": {
160
+ "intermediate_size": 11008,
161
+ "dead_count": 0,
162
+ "dead_fraction": 0.0,
163
+ "mean_activation_freq": 0.9766754339439001,
164
+ "min_activation_freq": 0.9200075092940734
165
+ },
166
+ "layer_20": {
167
+ "intermediate_size": 11008,
168
+ "dead_count": 0,
169
+ "dead_fraction": 0.0,
170
+ "mean_activation_freq": 0.9794000657739632,
171
+ "min_activation_freq": 0.9129820887471641
172
+ },
173
+ "layer_21": {
174
+ "intermediate_size": 11008,
175
+ "dead_count": 0,
176
+ "dead_fraction": 0.0,
177
+ "mean_activation_freq": 0.9795396099947247,
178
+ "min_activation_freq": 0.9153247769686778
179
+ },
180
+ "layer_22": {
181
+ "intermediate_size": 11008,
182
+ "dead_count": 0,
183
+ "dead_fraction": 0.0,
184
+ "mean_activation_freq": 0.9790669782916867,
185
+ "min_activation_freq": 0.9266495327844144
186
+ },
187
+ "layer_23": {
188
+ "intermediate_size": 11008,
189
+ "dead_count": 0,
190
+ "dead_fraction": 0.0,
191
+ "mean_activation_freq": 0.9793198090694186,
192
+ "min_activation_freq": 0.9360070650823114
193
+ },
194
+ "layer_24": {
195
+ "intermediate_size": 11008,
196
+ "dead_count": 0,
197
+ "dead_fraction": 0.0,
198
+ "mean_activation_freq": 0.9811667343626113,
199
+ "min_activation_freq": 0.934806635677608
200
+ },
201
+ "layer_25": {
202
+ "intermediate_size": 11008,
203
+ "dead_count": 0,
204
+ "dead_fraction": 0.0,
205
+ "mean_activation_freq": 0.9832285650722575,
206
+ "min_activation_freq": 0.9444338679739184
207
+ },
208
+ "layer_26": {
209
+ "intermediate_size": 11008,
210
+ "dead_count": 0,
211
+ "dead_fraction": 0.0,
212
+ "mean_activation_freq": 0.9853880129153925,
213
+ "min_activation_freq": 0.9525063591029037
214
+ },
215
+ "layer_27": {
216
+ "intermediate_size": 11008,
217
+ "dead_count": 0,
218
+ "dead_fraction": 0.0,
219
+ "mean_activation_freq": 0.9867653589625655,
220
+ "min_activation_freq": 0.9368293856657095
221
+ },
222
+ "layer_28": {
223
+ "intermediate_size": 11008,
224
+ "dead_count": 0,
225
+ "dead_fraction": 0.0,
226
+ "mean_activation_freq": 0.9870408826012386,
227
+ "min_activation_freq": 0.23825086330440667
228
+ },
229
+ "layer_29": {
230
+ "intermediate_size": 11008,
231
+ "dead_count": 0,
232
+ "dead_fraction": 0.0,
233
+ "mean_activation_freq": 0.9884186557233711,
234
+ "min_activation_freq": 0.9601452149403222
235
+ },
236
+ "layer_30": {
237
+ "intermediate_size": 11008,
238
+ "dead_count": 0,
239
+ "dead_fraction": 0.0,
240
+ "mean_activation_freq": 0.9900151282152624,
241
+ "min_activation_freq": 0.5783531377743932
242
+ },
243
+ "layer_31": {
244
+ "intermediate_size": 11008,
245
+ "dead_count": 0,
246
+ "dead_fraction": 0.0,
247
+ "mean_activation_freq": 0.9908374946767916,
248
+ "min_activation_freq": 0.9335982739200102
249
+ },
250
+ "layer_32": {
251
+ "intermediate_size": 11008,
252
+ "dead_count": 0,
253
+ "dead_fraction": 0.0,
254
+ "mean_activation_freq": 0.9890046842350142,
255
+ "min_activation_freq": 0.9203935504682732
256
+ },
257
+ "layer_33": {
258
+ "intermediate_size": 11008,
259
+ "dead_count": 0,
260
+ "dead_fraction": 0.0,
261
+ "mean_activation_freq": 0.9866693028865339,
262
+ "min_activation_freq": 0.9376014680141089
263
+ },
264
+ "layer_34": {
265
+ "intermediate_size": 11008,
266
+ "dead_count": 0,
267
+ "dead_fraction": 0.0,
268
+ "mean_activation_freq": 0.9879030645550098,
269
+ "min_activation_freq": 0.8934182623916572
270
+ },
271
+ "layer_35": {
272
+ "intermediate_size": 11008,
273
+ "dead_count": 0,
274
+ "dead_fraction": 0.0,
275
+ "mean_activation_freq": 0.9914988694917494,
276
+ "min_activation_freq": 0.006642023490341038
277
+ },
278
+ "_overall": {
279
+ "dead_count": 16439,
280
+ "intermediate_total": 396288,
281
+ "dead_fraction": 0.041482457202842375
282
+ }
283
+ },
284
+ "size": 770,
285
+ "size_pre_filter": 770,
286
+ "skipped_long": 0
287
+ },
288
+ "negative": {
289
+ "loss": 0.052044475198645995,
290
+ "token_entropy": 0.014608431552291079,
291
+ "num_supervised_tokens": 149231,
292
+ "kl_from_init": 3.393076465960805,
293
+ "kl_to_init": 0.3196539023368723,
294
+ "dead_units": {
295
+ "layer_0": {
296
+ "intermediate_size": 11008,
297
+ "dead_count": 0,
298
+ "dead_fraction": 0.0,
299
+ "mean_activation_freq": 0.9230990210295946,
300
+ "min_activation_freq": 0.7728689079346784
301
+ },
302
+ "layer_1": {
303
+ "intermediate_size": 11008,
304
+ "dead_count": 8853,
305
+ "dead_fraction": 0.8042332848837209,
306
+ "mean_activation_freq": 0.14579925865187465,
307
+ "min_activation_freq": 0.0
308
+ },
309
+ "layer_2": {
310
+ "intermediate_size": 11008,
311
+ "dead_count": 7670,
312
+ "dead_fraction": 0.696765988372093,
313
+ "mean_activation_freq": 0.16607041807881587,
314
+ "min_activation_freq": 0.0
315
+ },
316
+ "layer_3": {
317
+ "intermediate_size": 11008,
318
+ "dead_count": 335,
319
+ "dead_fraction": 0.030432412790697673,
320
+ "mean_activation_freq": 0.38562429948524474,
321
+ "min_activation_freq": 0.0
322
+ },
323
+ "layer_4": {
324
+ "intermediate_size": 11008,
325
+ "dead_count": 34,
326
+ "dead_fraction": 0.0030886627906976743,
327
+ "mean_activation_freq": 0.5614262107624078,
328
+ "min_activation_freq": 0.0
329
+ },
330
+ "layer_5": {
331
+ "intermediate_size": 11008,
332
+ "dead_count": 32,
333
+ "dead_fraction": 0.0029069767441860465,
334
+ "mean_activation_freq": 0.7232661195728162,
335
+ "min_activation_freq": 0.0
336
+ },
337
+ "layer_6": {
338
+ "intermediate_size": 11008,
339
+ "dead_count": 0,
340
+ "dead_fraction": 0.0,
341
+ "mean_activation_freq": 0.8749037105713118,
342
+ "min_activation_freq": 0.13302195924439292
343
+ },
344
+ "layer_7": {
345
+ "intermediate_size": 11008,
346
+ "dead_count": 0,
347
+ "dead_fraction": 0.0,
348
+ "mean_activation_freq": 0.8873730759256409,
349
+ "min_activation_freq": 0.05696537582673841
350
+ },
351
+ "layer_8": {
352
+ "intermediate_size": 11008,
353
+ "dead_count": 0,
354
+ "dead_fraction": 0.0,
355
+ "mean_activation_freq": 0.9424713488515465,
356
+ "min_activation_freq": 0.06736535974428906
357
+ },
358
+ "layer_9": {
359
+ "intermediate_size": 11008,
360
+ "dead_count": 0,
361
+ "dead_fraction": 0.0,
362
+ "mean_activation_freq": 0.9509299549479089,
363
+ "min_activation_freq": 0.2745676166480155
364
+ },
365
+ "layer_10": {
366
+ "intermediate_size": 11008,
367
+ "dead_count": 0,
368
+ "dead_fraction": 0.0,
369
+ "mean_activation_freq": 0.9740962985890222,
370
+ "min_activation_freq": 0.6655721666409794
371
+ },
372
+ "layer_11": {
373
+ "intermediate_size": 11008,
374
+ "dead_count": 0,
375
+ "dead_fraction": 0.0,
376
+ "mean_activation_freq": 0.9852091881842152,
377
+ "min_activation_freq": 0.9322526820834813
378
+ },
379
+ "layer_12": {
380
+ "intermediate_size": 11008,
381
+ "dead_count": 0,
382
+ "dead_fraction": 0.0,
383
+ "mean_activation_freq": 0.9866892000089841,
384
+ "min_activation_freq": 0.911064055055585
385
+ },
386
+ "layer_13": {
387
+ "intermediate_size": 11008,
388
+ "dead_count": 0,
389
+ "dead_fraction": 0.0,
390
+ "mean_activation_freq": 0.9802371204095868,
391
+ "min_activation_freq": 0.9101125101352936
392
+ },
393
+ "layer_14": {
394
+ "intermediate_size": 11008,
395
+ "dead_count": 0,
396
+ "dead_fraction": 0.0,
397
+ "mean_activation_freq": 0.9826654112593586,
398
+ "min_activation_freq": 0.8715481367812318
399
+ },
400
+ "layer_15": {
401
+ "intermediate_size": 11008,
402
+ "dead_count": 0,
403
+ "dead_fraction": 0.0,
404
+ "mean_activation_freq": 0.9804418375618461,
405
+ "min_activation_freq": 0.9047784977652097
406
+ },
407
+ "layer_16": {
408
+ "intermediate_size": 11008,
409
+ "dead_count": 0,
410
+ "dead_fraction": 0.0,
411
+ "mean_activation_freq": 0.9793969744775269,
412
+ "min_activation_freq": 0.9038403548860492
413
+ },
414
+ "layer_17": {
415
+ "intermediate_size": 11008,
416
+ "dead_count": 0,
417
+ "dead_fraction": 0.0,
418
+ "mean_activation_freq": 0.9788950535491411,
419
+ "min_activation_freq": 0.9139052877753282
420
+ },
421
+ "layer_18": {
422
+ "intermediate_size": 11008,
423
+ "dead_count": 0,
424
+ "dead_fraction": 0.0,
425
+ "mean_activation_freq": 0.9796025846527852,
426
+ "min_activation_freq": 0.9164851806930195
427
+ },
428
+ "layer_19": {
429
+ "intermediate_size": 11008,
430
+ "dead_count": 0,
431
+ "dead_fraction": 0.0,
432
+ "mean_activation_freq": 0.9766960115038603,
433
+ "min_activation_freq": 0.9188908470760097
434
+ },
435
+ "layer_20": {
436
+ "intermediate_size": 11008,
437
+ "dead_count": 0,
438
+ "dead_fraction": 0.0,
439
+ "mean_activation_freq": 0.9794454643479994,
440
+ "min_activation_freq": 0.914414565338301
441
+ },
442
+ "layer_21": {
443
+ "intermediate_size": 11008,
444
+ "dead_count": 0,
445
+ "dead_fraction": 0.0,
446
+ "mean_activation_freq": 0.9794191229088725,
447
+ "min_activation_freq": 0.9139253908370245
448
+ },
449
+ "layer_22": {
450
+ "intermediate_size": 11008,
451
+ "dead_count": 0,
452
+ "dead_fraction": 0.0,
453
+ "mean_activation_freq": 0.9788112761822899,
454
+ "min_activation_freq": 0.9266305258290838
455
+ },
456
+ "layer_23": {
457
+ "intermediate_size": 11008,
458
+ "dead_count": 0,
459
+ "dead_fraction": 0.0,
460
+ "mean_activation_freq": 0.978984599345396,
461
+ "min_activation_freq": 0.9339145351837085
462
+ },
463
+ "layer_24": {
464
+ "intermediate_size": 11008,
465
+ "dead_count": 0,
466
+ "dead_fraction": 0.0,
467
+ "mean_activation_freq": 0.9809927396146648,
468
+ "min_activation_freq": 0.933907834163143
469
+ },
470
+ "layer_25": {
471
+ "intermediate_size": 11008,
472
+ "dead_count": 0,
473
+ "dead_fraction": 0.0,
474
+ "mean_activation_freq": 0.9830062702851968,
475
+ "min_activation_freq": 0.9445490548210492
476
+ },
477
+ "layer_26": {
478
+ "intermediate_size": 11008,
479
+ "dead_count": 0,
480
+ "dead_fraction": 0.0,
481
+ "mean_activation_freq": 0.9852088296357856,
482
+ "min_activation_freq": 0.9504191488363678
483
+ },
484
+ "layer_27": {
485
+ "intermediate_size": 11008,
486
+ "dead_count": 0,
487
+ "dead_fraction": 0.0,
488
+ "mean_activation_freq": 0.9866468985991829,
489
+ "min_activation_freq": 0.9364408199368764
490
+ },
491
+ "layer_28": {
492
+ "intermediate_size": 11008,
493
+ "dead_count": 0,
494
+ "dead_fraction": 0.0,
495
+ "mean_activation_freq": 0.9869748322280674,
496
+ "min_activation_freq": 0.24478828125523516
497
+ },
498
+ "layer_29": {
499
+ "intermediate_size": 11008,
500
+ "dead_count": 0,
501
+ "dead_fraction": 0.0,
502
+ "mean_activation_freq": 0.9883655332305947,
503
+ "min_activation_freq": 0.959867587833627
504
+ },
505
+ "layer_30": {
506
+ "intermediate_size": 11008,
507
+ "dead_count": 0,
508
+ "dead_fraction": 0.0,
509
+ "mean_activation_freq": 0.9899868453998973,
510
+ "min_activation_freq": 0.6061207121844657
511
+ },
512
+ "layer_31": {
513
+ "intermediate_size": 11008,
514
+ "dead_count": 0,
515
+ "dead_fraction": 0.0,
516
+ "mean_activation_freq": 0.9908114629586282,
517
+ "min_activation_freq": 0.9333985566001702
518
+ },
519
+ "layer_32": {
520
+ "intermediate_size": 11008,
521
+ "dead_count": 0,
522
+ "dead_fraction": 0.0,
523
+ "mean_activation_freq": 0.9889625011473548,
524
+ "min_activation_freq": 0.9172289939757825
525
+ },
526
+ "layer_33": {
527
+ "intermediate_size": 11008,
528
+ "dead_count": 0,
529
+ "dead_fraction": 0.0,
530
+ "mean_activation_freq": 0.9865908536445682,
531
+ "min_activation_freq": 0.9362866964638714
532
+ },
533
+ "layer_34": {
534
+ "intermediate_size": 11008,
535
+ "dead_count": 0,
536
+ "dead_fraction": 0.0,
537
+ "mean_activation_freq": 0.9877843578807429,
538
+ "min_activation_freq": 0.8999202578552714
539
+ },
540
+ "layer_35": {
541
+ "intermediate_size": 11008,
542
+ "dead_count": 0,
543
+ "dead_fraction": 0.0,
544
+ "mean_activation_freq": 0.9917539717280046,
545
+ "min_activation_freq": 0.006419577701683966
546
+ },
547
+ "_overall": {
548
+ "dead_count": 16924,
549
+ "intermediate_total": 396288,
550
+ "dead_fraction": 0.0427063145994832
551
+ }
552
+ },
553
+ "size": 254,
554
+ "size_pre_filter": 254,
555
+ "skipped_long": 0
556
+ }
557
+ },
558
+ "old_data": {
559
+ "num_records": 56320,
560
+ "source_steps": [
561
+ 25,
562
+ 50,
563
+ 75,
564
+ 100,
565
+ 125,
566
+ 150,
567
+ 175,
568
+ 200,
569
+ 225,
570
+ 250,
571
+ 275,
572
+ 300,
573
+ 325,
574
+ 350,
575
+ 375,
576
+ 400,
577
+ 425,
578
+ 450,
579
+ 475,
580
+ 500,
581
+ 525,
582
+ 550,
583
+ 575,
584
+ 600,
585
+ 625,
586
+ 650,
587
+ 675,
588
+ 700,
589
+ 725,
590
+ 750,
591
+ 775,
592
+ 800,
593
+ 825,
594
+ 850,
595
+ 875,
596
+ 900,
597
+ 925,
598
+ 950,
599
+ 975,
600
+ 1000,
601
+ 1025,
602
+ 1050,
603
+ 1075,
604
+ 1100,
605
+ 1125,
606
+ 1150,
607
+ 1175,
608
+ 1200,
609
+ 1225,
610
+ 1250,
611
+ 1275,
612
+ 1300,
613
+ 1325,
614
+ 1350,
615
+ 1375
616
+ ],
617
+ "num_source_steps": 55,
618
+ "num_positives_pre_cap": 29632,
619
+ "num_negatives_pre_cap": 26688,
620
+ "positive": {
621
+ "loss": 0.5201052261762237,
622
+ "token_entropy": 0.01571655626127166,
623
+ "num_supervised_tokens": 12282905,
624
+ "kl_from_init": 3.6794296907235635,
625
+ "kl_to_init": 0.3542914716685758,
626
+ "dead_units": {
627
+ "layer_0": {
628
+ "intermediate_size": 11008,
629
+ "dead_count": 0,
630
+ "dead_fraction": 0.0,
631
+ "mean_activation_freq": 0.9230721695416738,
632
+ "min_activation_freq": 0.7774596481858324
633
+ },
634
+ "layer_1": {
635
+ "intermediate_size": 11008,
636
+ "dead_count": 8725,
637
+ "dead_fraction": 0.7926053779069767,
638
+ "mean_activation_freq": 0.1463331459969989,
639
+ "min_activation_freq": 0.0
640
+ },
641
+ "layer_2": {
642
+ "intermediate_size": 11008,
643
+ "dead_count": 3065,
644
+ "dead_fraction": 0.27843386627906974,
645
+ "mean_activation_freq": 0.1664109302971568,
646
+ "min_activation_freq": 0.0
647
+ },
648
+ "layer_3": {
649
+ "intermediate_size": 11008,
650
+ "dead_count": 36,
651
+ "dead_fraction": 0.0032703488372093025,
652
+ "mean_activation_freq": 0.38799552035247126,
653
+ "min_activation_freq": 0.0
654
+ },
655
+ "layer_4": {
656
+ "intermediate_size": 11008,
657
+ "dead_count": 1,
658
+ "dead_fraction": 9.084302325581395e-05,
659
+ "mean_activation_freq": 0.5592902087826702,
660
+ "min_activation_freq": 0.0
661
+ },
662
+ "layer_5": {
663
+ "intermediate_size": 11008,
664
+ "dead_count": 0,
665
+ "dead_fraction": 0.0,
666
+ "mean_activation_freq": 0.7210712271460349,
667
+ "min_activation_freq": 1.953935164360548e-06
668
+ },
669
+ "layer_6": {
670
+ "intermediate_size": 11008,
671
+ "dead_count": 0,
672
+ "dead_fraction": 0.0,
673
+ "mean_activation_freq": 0.8737310338432007,
674
+ "min_activation_freq": 0.12853351874007005
675
+ },
676
+ "layer_7": {
677
+ "intermediate_size": 11008,
678
+ "dead_count": 0,
679
+ "dead_fraction": 0.0,
680
+ "mean_activation_freq": 0.8832260707103197,
681
+ "min_activation_freq": 0.05619167452650656
682
+ },
683
+ "layer_8": {
684
+ "intermediate_size": 11008,
685
+ "dead_count": 0,
686
+ "dead_fraction": 0.0,
687
+ "mean_activation_freq": 0.942279175563884,
688
+ "min_activation_freq": 0.07650893660742308
689
+ },
690
+ "layer_9": {
691
+ "intermediate_size": 11008,
692
+ "dead_count": 0,
693
+ "dead_fraction": 0.0,
694
+ "mean_activation_freq": 0.9503062967765241,
695
+ "min_activation_freq": 0.2519108468232882
696
+ },
697
+ "layer_10": {
698
+ "intermediate_size": 11008,
699
+ "dead_count": 0,
700
+ "dead_fraction": 0.0,
701
+ "mean_activation_freq": 0.9738468789561212,
702
+ "min_activation_freq": 0.6677344650960013
703
+ },
704
+ "layer_11": {
705
+ "intermediate_size": 11008,
706
+ "dead_count": 0,
707
+ "dead_fraction": 0.0,
708
+ "mean_activation_freq": 0.9852253442084232,
709
+ "min_activation_freq": 0.9324500189491004
710
+ },
711
+ "layer_12": {
712
+ "intermediate_size": 11008,
713
+ "dead_count": 0,
714
+ "dead_fraction": 0.0,
715
+ "mean_activation_freq": 0.9867091128881237,
716
+ "min_activation_freq": 0.9136281685806412
717
+ },
718
+ "layer_13": {
719
+ "intermediate_size": 11008,
720
+ "dead_count": 0,
721
+ "dead_fraction": 0.0,
722
+ "mean_activation_freq": 0.9805796595318033,
723
+ "min_activation_freq": 0.9107969165274827
724
+ },
725
+ "layer_14": {
726
+ "intermediate_size": 11008,
727
+ "dead_count": 0,
728
+ "dead_fraction": 0.0,
729
+ "mean_activation_freq": 0.9828297139800549,
730
+ "min_activation_freq": 0.8775414285138573
731
+ },
732
+ "layer_15": {
733
+ "intermediate_size": 11008,
734
+ "dead_count": 0,
735
+ "dead_fraction": 0.0,
736
+ "mean_activation_freq": 0.9806388601980264,
737
+ "min_activation_freq": 0.9070140980492808
738
+ },
739
+ "layer_16": {
740
+ "intermediate_size": 11008,
741
+ "dead_count": 0,
742
+ "dead_fraction": 0.0,
743
+ "mean_activation_freq": 0.979579855635621,
744
+ "min_activation_freq": 0.909217078533132
745
+ },
746
+ "layer_17": {
747
+ "intermediate_size": 11008,
748
+ "dead_count": 0,
749
+ "dead_fraction": 0.0,
750
+ "mean_activation_freq": 0.9790229154355369,
751
+ "min_activation_freq": 0.9154161820839615
752
+ },
753
+ "layer_18": {
754
+ "intermediate_size": 11008,
755
+ "dead_count": 0,
756
+ "dead_fraction": 0.0,
757
+ "mean_activation_freq": 0.9796741825597692,
758
+ "min_activation_freq": 0.920506753084877
759
+ },
760
+ "layer_19": {
761
+ "intermediate_size": 11008,
762
+ "dead_count": 0,
763
+ "dead_fraction": 0.0,
764
+ "mean_activation_freq": 0.9767044930109566,
765
+ "min_activation_freq": 0.9211310353698902
766
+ },
767
+ "layer_20": {
768
+ "intermediate_size": 11008,
769
+ "dead_count": 0,
770
+ "dead_fraction": 0.0,
771
+ "mean_activation_freq": 0.9793659415588857,
772
+ "min_activation_freq": 0.9104395906343004
773
+ },
774
+ "layer_21": {
775
+ "intermediate_size": 11008,
776
+ "dead_count": 0,
777
+ "dead_fraction": 0.0,
778
+ "mean_activation_freq": 0.9795575629861509,
779
+ "min_activation_freq": 0.9173554627345893
780
+ },
781
+ "layer_22": {
782
+ "intermediate_size": 11008,
783
+ "dead_count": 0,
784
+ "dead_fraction": 0.0,
785
+ "mean_activation_freq": 0.9791860162372887,
786
+ "min_activation_freq": 0.930643117405858
787
+ },
788
+ "layer_23": {
789
+ "intermediate_size": 11008,
790
+ "dead_count": 0,
791
+ "dead_fraction": 0.0,
792
+ "mean_activation_freq": 0.9795332508961123,
793
+ "min_activation_freq": 0.9369103644455445
794
+ },
795
+ "layer_24": {
796
+ "intermediate_size": 11008,
797
+ "dead_count": 0,
798
+ "dead_fraction": 0.0,
799
+ "mean_activation_freq": 0.9813089986844474,
800
+ "min_activation_freq": 0.9331517259150014
801
+ },
802
+ "layer_25": {
803
+ "intermediate_size": 11008,
804
+ "dead_count": 0,
805
+ "dead_fraction": 0.0,
806
+ "mean_activation_freq": 0.9833674699200012,
807
+ "min_activation_freq": 0.9462714235761004
808
+ },
809
+ "layer_26": {
810
+ "intermediate_size": 11008,
811
+ "dead_count": 0,
812
+ "dead_fraction": 0.0,
813
+ "mean_activation_freq": 0.9854754681372224,
814
+ "min_activation_freq": 0.953226130137781
815
+ },
816
+ "layer_27": {
817
+ "intermediate_size": 11008,
818
+ "dead_count": 0,
819
+ "dead_fraction": 0.0,
820
+ "mean_activation_freq": 0.9868525726003592,
821
+ "min_activation_freq": 0.9372785998100612
822
+ },
823
+ "layer_28": {
824
+ "intermediate_size": 11008,
825
+ "dead_count": 0,
826
+ "dead_fraction": 0.0,
827
+ "mean_activation_freq": 0.9871128639707757,
828
+ "min_activation_freq": 0.25365383840386296
829
+ },
830
+ "layer_29": {
831
+ "intermediate_size": 11008,
832
+ "dead_count": 0,
833
+ "dead_fraction": 0.0,
834
+ "mean_activation_freq": 0.988481737369615,
835
+ "min_activation_freq": 0.958784750024526
836
+ },
837
+ "layer_30": {
838
+ "intermediate_size": 11008,
839
+ "dead_count": 0,
840
+ "dead_fraction": 0.0,
841
+ "mean_activation_freq": 0.9900673516285865,
842
+ "min_activation_freq": 0.5749700905445414
843
+ },
844
+ "layer_31": {
845
+ "intermediate_size": 11008,
846
+ "dead_count": 0,
847
+ "dead_fraction": 0.0,
848
+ "mean_activation_freq": 0.9908985027239906,
849
+ "min_activation_freq": 0.9395348250271415
850
+ },
851
+ "layer_32": {
852
+ "intermediate_size": 11008,
853
+ "dead_count": 0,
854
+ "dead_fraction": 0.0,
855
+ "mean_activation_freq": 0.9891065894044667,
856
+ "min_activation_freq": 0.9166418693297718
857
+ },
858
+ "layer_33": {
859
+ "intermediate_size": 11008,
860
+ "dead_count": 0,
861
+ "dead_fraction": 0.0,
862
+ "mean_activation_freq": 0.9868158620464925,
863
+ "min_activation_freq": 0.9373382762465394
864
+ },
865
+ "layer_34": {
866
+ "intermediate_size": 11008,
867
+ "dead_count": 0,
868
+ "dead_fraction": 0.0,
869
+ "mean_activation_freq": 0.988016199470044,
870
+ "min_activation_freq": 0.8897006042137426
871
+ },
872
+ "layer_35": {
873
+ "intermediate_size": 11008,
874
+ "dead_count": 0,
875
+ "dead_fraction": 0.0,
876
+ "mean_activation_freq": 0.9914204574247422,
877
+ "min_activation_freq": 0.007052077664037946
878
+ },
879
+ "_overall": {
880
+ "dead_count": 11827,
881
+ "intermediate_total": 396288,
882
+ "dead_fraction": 0.029844456556847546
883
+ }
884
+ },
885
+ "size": 29632,
886
+ "size_pre_filter": 29632,
887
+ "skipped_long": 0
888
+ },
889
+ "negative": {
890
+ "loss": 1.0415867224022055,
891
+ "token_entropy": 0.030672794999767564,
892
+ "num_supervised_tokens": 12445165,
893
+ "kl_from_init": 3.428990164864554,
894
+ "kl_to_init": 0.3800119474408492,
895
+ "dead_units": {
896
+ "layer_0": {
897
+ "intermediate_size": 11008,
898
+ "dead_count": 0,
899
+ "dead_fraction": 0.0,
900
+ "mean_activation_freq": 0.9233862426223333,
901
+ "min_activation_freq": 0.7796547494549089
902
+ },
903
+ "layer_1": {
904
+ "intermediate_size": 11008,
905
+ "dead_count": 4976,
906
+ "dead_fraction": 0.45203488372093026,
907
+ "mean_activation_freq": 0.14643260602121014,
908
+ "min_activation_freq": 0.0
909
+ },
910
+ "layer_2": {
911
+ "intermediate_size": 11008,
912
+ "dead_count": 832,
913
+ "dead_fraction": 0.0755813953488372,
914
+ "mean_activation_freq": 0.1664401887736004,
915
+ "min_activation_freq": 0.0
916
+ },
917
+ "layer_3": {
918
+ "intermediate_size": 11008,
919
+ "dead_count": 0,
920
+ "dead_fraction": 0.0,
921
+ "mean_activation_freq": 0.3880659475606778,
922
+ "min_activation_freq": 8.035249030446764e-07
923
+ },
924
+ "layer_4": {
925
+ "intermediate_size": 11008,
926
+ "dead_count": 0,
927
+ "dead_fraction": 0.0,
928
+ "mean_activation_freq": 0.5601628403786965,
929
+ "min_activation_freq": 3.133747121874238e-06
930
+ },
931
+ "layer_5": {
932
+ "intermediate_size": 11008,
933
+ "dead_count": 0,
934
+ "dead_fraction": 0.0,
935
+ "mean_activation_freq": 0.72094876515782,
936
+ "min_activation_freq": 2.860548654839048e-05
937
+ },
938
+ "layer_6": {
939
+ "intermediate_size": 11008,
940
+ "dead_count": 0,
941
+ "dead_fraction": 0.0,
942
+ "mean_activation_freq": 0.8727890060077759,
943
+ "min_activation_freq": 0.12498862007856064
944
+ },
945
+ "layer_7": {
946
+ "intermediate_size": 11008,
947
+ "dead_count": 0,
948
+ "dead_fraction": 0.0,
949
+ "mean_activation_freq": 0.8844133488099549,
950
+ "min_activation_freq": 0.05985457002779795
951
+ },
952
+ "layer_8": {
953
+ "intermediate_size": 11008,
954
+ "dead_count": 0,
955
+ "dead_fraction": 0.0,
956
+ "mean_activation_freq": 0.9423915219000736,
957
+ "min_activation_freq": 0.07712577535131114
958
+ },
959
+ "layer_9": {
960
+ "intermediate_size": 11008,
961
+ "dead_count": 0,
962
+ "dead_fraction": 0.0,
963
+ "mean_activation_freq": 0.9507491773779496,
964
+ "min_activation_freq": 0.26087046656271734
965
+ },
966
+ "layer_10": {
967
+ "intermediate_size": 11008,
968
+ "dead_count": 0,
969
+ "dead_fraction": 0.0,
970
+ "mean_activation_freq": 0.9739478586961795,
971
+ "min_activation_freq": 0.6779456118098877
972
+ },
973
+ "layer_11": {
974
+ "intermediate_size": 11008,
975
+ "dead_count": 0,
976
+ "dead_fraction": 0.0,
977
+ "mean_activation_freq": 0.9852095064086437,
978
+ "min_activation_freq": 0.9323300253552284
979
+ },
980
+ "layer_12": {
981
+ "intermediate_size": 11008,
982
+ "dead_count": 0,
983
+ "dead_fraction": 0.0,
984
+ "mean_activation_freq": 0.9867071567371513,
985
+ "min_activation_freq": 0.9143284158948476
986
+ },
987
+ "layer_13": {
988
+ "intermediate_size": 11008,
989
+ "dead_count": 0,
990
+ "dead_fraction": 0.0,
991
+ "mean_activation_freq": 0.9805636472253809,
992
+ "min_activation_freq": 0.9105158509348812
993
+ },
994
+ "layer_14": {
995
+ "intermediate_size": 11008,
996
+ "dead_count": 0,
997
+ "dead_fraction": 0.0,
998
+ "mean_activation_freq": 0.9827834868749323,
999
+ "min_activation_freq": 0.8798892581978625
1000
+ },
1001
+ "layer_15": {
1002
+ "intermediate_size": 11008,
1003
+ "dead_count": 0,
1004
+ "dead_fraction": 0.0,
1005
+ "mean_activation_freq": 0.9806291603450786,
1006
+ "min_activation_freq": 0.9099158588897777
1007
+ },
1008
+ "layer_16": {
1009
+ "intermediate_size": 11008,
1010
+ "dead_count": 0,
1011
+ "dead_fraction": 0.0,
1012
+ "mean_activation_freq": 0.9796107430427544,
1013
+ "min_activation_freq": 0.9060154686579085
1014
+ },
1015
+ "layer_17": {
1016
+ "intermediate_size": 11008,
1017
+ "dead_count": 0,
1018
+ "dead_fraction": 0.0,
1019
+ "mean_activation_freq": 0.9790326464600025,
1020
+ "min_activation_freq": 0.9182766158584479
1021
+ },
1022
+ "layer_18": {
1023
+ "intermediate_size": 11008,
1024
+ "dead_count": 0,
1025
+ "dead_fraction": 0.0,
1026
+ "mean_activation_freq": 0.9796877444444219,
1027
+ "min_activation_freq": 0.9211512261990902
1028
+ },
1029
+ "layer_19": {
1030
+ "intermediate_size": 11008,
1031
+ "dead_count": 0,
1032
+ "dead_fraction": 0.0,
1033
+ "mean_activation_freq": 0.9766979679554199,
1034
+ "min_activation_freq": 0.9223987789635574
1035
+ },
1036
+ "layer_20": {
1037
+ "intermediate_size": 11008,
1038
+ "dead_count": 0,
1039
+ "dead_fraction": 0.0,
1040
+ "mean_activation_freq": 0.9793531411123387,
1041
+ "min_activation_freq": 0.9136692040643897
1042
+ },
1043
+ "layer_21": {
1044
+ "intermediate_size": 11008,
1045
+ "dead_count": 0,
1046
+ "dead_fraction": 0.0,
1047
+ "mean_activation_freq": 0.9794186047343768,
1048
+ "min_activation_freq": 0.917518168702464
1049
+ },
1050
+ "layer_22": {
1051
+ "intermediate_size": 11008,
1052
+ "dead_count": 0,
1053
+ "dead_fraction": 0.0,
1054
+ "mean_activation_freq": 0.9789575730582362,
1055
+ "min_activation_freq": 0.9306250258634579
1056
+ },
1057
+ "layer_23": {
1058
+ "intermediate_size": 11008,
1059
+ "dead_count": 0,
1060
+ "dead_fraction": 0.0,
1061
+ "mean_activation_freq": 0.9792072060333457,
1062
+ "min_activation_freq": 0.9353696797109561
1063
+ },
1064
+ "layer_24": {
1065
+ "intermediate_size": 11008,
1066
+ "dead_count": 0,
1067
+ "dead_fraction": 0.0,
1068
+ "mean_activation_freq": 0.9811054457093541,
1069
+ "min_activation_freq": 0.9322626899683532
1070
+ },
1071
+ "layer_25": {
1072
+ "intermediate_size": 11008,
1073
+ "dead_count": 0,
1074
+ "dead_fraction": 0.0,
1075
+ "mean_activation_freq": 0.9831177583369236,
1076
+ "min_activation_freq": 0.945492004324571
1077
+ },
1078
+ "layer_26": {
1079
+ "intermediate_size": 11008,
1080
+ "dead_count": 0,
1081
+ "dead_fraction": 0.0,
1082
+ "mean_activation_freq": 0.9852768825046249,
1083
+ "min_activation_freq": 0.9536628883586518
1084
+ },
1085
+ "layer_27": {
1086
+ "intermediate_size": 11008,
1087
+ "dead_count": 0,
1088
+ "dead_fraction": 0.0,
1089
+ "mean_activation_freq": 0.9867168267009531,
1090
+ "min_activation_freq": 0.9353044334888289
1091
+ },
1092
+ "layer_28": {
1093
+ "intermediate_size": 11008,
1094
+ "dead_count": 0,
1095
+ "dead_fraction": 0.0,
1096
+ "mean_activation_freq": 0.987016358557943,
1097
+ "min_activation_freq": 0.272265012155323
1098
+ },
1099
+ "layer_29": {
1100
+ "intermediate_size": 11008,
1101
+ "dead_count": 0,
1102
+ "dead_fraction": 0.0,
1103
+ "mean_activation_freq": 0.9883948725892839,
1104
+ "min_activation_freq": 0.9589663937762175
1105
+ },
1106
+ "layer_30": {
1107
+ "intermediate_size": 11008,
1108
+ "dead_count": 0,
1109
+ "dead_fraction": 0.0,
1110
+ "mean_activation_freq": 0.9900057847238101,
1111
+ "min_activation_freq": 0.5906296943431445
1112
+ },
1113
+ "layer_31": {
1114
+ "intermediate_size": 11008,
1115
+ "dead_count": 0,
1116
+ "dead_fraction": 0.0,
1117
+ "mean_activation_freq": 0.9908837665013651,
1118
+ "min_activation_freq": 0.9405123194429323
1119
+ },
1120
+ "layer_32": {
1121
+ "intermediate_size": 11008,
1122
+ "dead_count": 0,
1123
+ "dead_fraction": 0.0,
1124
+ "mean_activation_freq": 0.9891073056887699,
1125
+ "min_activation_freq": 0.9121621127562392
1126
+ },
1127
+ "layer_33": {
1128
+ "intermediate_size": 11008,
1129
+ "dead_count": 0,
1130
+ "dead_fraction": 0.0,
1131
+ "mean_activation_freq": 0.9868587922077967,
1132
+ "min_activation_freq": 0.9359413876794723
1133
+ },
1134
+ "layer_34": {
1135
+ "intermediate_size": 11008,
1136
+ "dead_count": 0,
1137
+ "dead_fraction": 0.0,
1138
+ "mean_activation_freq": 0.9879958582980423,
1139
+ "min_activation_freq": 0.9043002643998694
1140
+ },
1141
+ "layer_35": {
1142
+ "intermediate_size": 11008,
1143
+ "dead_count": 0,
1144
+ "dead_fraction": 0.0,
1145
+ "mean_activation_freq": 0.991780657793679,
1146
+ "min_activation_freq": 0.014907476116226663
1147
+ },
1148
+ "_overall": {
1149
+ "dead_count": 5808,
1150
+ "intermediate_total": 396288,
1151
+ "dead_fraction": 0.014656007751937985
1152
+ }
1153
+ },
1154
+ "size": 26688,
1155
+ "size_pre_filter": 26688,
1156
+ "skipped_long": 0
1157
+ }
1158
+ }
1159
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/generation_config.json ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token_id": 151643,
3
+ "eos_token_id": 151643,
4
+ "max_new_tokens": 2048,
5
+ "transformers_version": "4.57.6"
6
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/model-00001-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6e18f65fd6f794db89b49c45a3eb84d3aa61b134a6249d894100eee17a7a1270
3
+ size 4987423696
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/model-00002-of-00002.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8ca5394b027b9c1b632a2be390a13f89570626f910187cf456b48b81d61f48c0
3
+ size 1806833320
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/model.safetensors.index.json ADDED
@@ -0,0 +1,443 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_parameters": 3397103616,
4
+ "total_size": 6794207232
5
+ },
6
+ "weight_map": {
7
+ "lm_head.weight": "model-00001-of-00002.safetensors",
8
+ "model.embed_tokens.weight": "model-00001-of-00002.safetensors",
9
+ "model.layers.0.input_layernorm.weight": "model-00002-of-00002.safetensors",
10
+ "model.layers.0.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
11
+ "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
12
+ "model.layers.0.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
13
+ "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
14
+ "model.layers.0.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
15
+ "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
16
+ "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
17
+ "model.layers.0.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
18
+ "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
19
+ "model.layers.0.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
20
+ "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
21
+ "model.layers.1.input_layernorm.weight": "model-00002-of-00002.safetensors",
22
+ "model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
23
+ "model.layers.1.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
24
+ "model.layers.1.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
25
+ "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
26
+ "model.layers.1.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
27
+ "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
28
+ "model.layers.1.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
29
+ "model.layers.1.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
30
+ "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
31
+ "model.layers.1.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
32
+ "model.layers.1.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
33
+ "model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors",
34
+ "model.layers.10.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
35
+ "model.layers.10.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
36
+ "model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
37
+ "model.layers.10.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
38
+ "model.layers.10.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
39
+ "model.layers.10.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
40
+ "model.layers.10.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
41
+ "model.layers.10.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
42
+ "model.layers.10.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
43
+ "model.layers.10.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
44
+ "model.layers.10.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
45
+ "model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
46
+ "model.layers.11.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
47
+ "model.layers.11.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
48
+ "model.layers.11.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
49
+ "model.layers.11.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
50
+ "model.layers.11.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
51
+ "model.layers.11.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
52
+ "model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
53
+ "model.layers.11.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
54
+ "model.layers.11.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
55
+ "model.layers.11.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
56
+ "model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
57
+ "model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
58
+ "model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
59
+ "model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
60
+ "model.layers.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
61
+ "model.layers.12.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
62
+ "model.layers.12.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
63
+ "model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
64
+ "model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
65
+ "model.layers.12.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
66
+ "model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
67
+ "model.layers.12.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
68
+ "model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
69
+ "model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
70
+ "model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
71
+ "model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
72
+ "model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
73
+ "model.layers.13.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
74
+ "model.layers.13.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
75
+ "model.layers.13.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
76
+ "model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
77
+ "model.layers.13.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
78
+ "model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
79
+ "model.layers.13.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
80
+ "model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
81
+ "model.layers.14.input_layernorm.weight": "model-00001-of-00002.safetensors",
82
+ "model.layers.14.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
83
+ "model.layers.14.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
84
+ "model.layers.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
85
+ "model.layers.14.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
86
+ "model.layers.14.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
87
+ "model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
88
+ "model.layers.14.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
89
+ "model.layers.14.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
90
+ "model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
91
+ "model.layers.14.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
92
+ "model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
93
+ "model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
94
+ "model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
95
+ "model.layers.15.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
96
+ "model.layers.15.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
97
+ "model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
98
+ "model.layers.15.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
99
+ "model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
100
+ "model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
101
+ "model.layers.15.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
102
+ "model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
103
+ "model.layers.15.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
104
+ "model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
105
+ "model.layers.16.input_layernorm.weight": "model-00002-of-00002.safetensors",
106
+ "model.layers.16.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
107
+ "model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
108
+ "model.layers.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
109
+ "model.layers.16.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
110
+ "model.layers.16.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
111
+ "model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
112
+ "model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
113
+ "model.layers.16.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
114
+ "model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
115
+ "model.layers.16.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
116
+ "model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
117
+ "model.layers.17.input_layernorm.weight": "model-00002-of-00002.safetensors",
118
+ "model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
119
+ "model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
120
+ "model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
121
+ "model.layers.17.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
122
+ "model.layers.17.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
123
+ "model.layers.17.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
124
+ "model.layers.17.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
125
+ "model.layers.17.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
126
+ "model.layers.17.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
127
+ "model.layers.17.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
128
+ "model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
129
+ "model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
130
+ "model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
131
+ "model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
132
+ "model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
133
+ "model.layers.18.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
134
+ "model.layers.18.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
135
+ "model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
136
+ "model.layers.18.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
137
+ "model.layers.18.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
138
+ "model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
139
+ "model.layers.18.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
140
+ "model.layers.18.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
141
+ "model.layers.19.input_layernorm.weight": "model-00001-of-00002.safetensors",
142
+ "model.layers.19.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
143
+ "model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
144
+ "model.layers.19.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
145
+ "model.layers.19.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
146
+ "model.layers.19.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
147
+ "model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
148
+ "model.layers.19.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
149
+ "model.layers.19.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
150
+ "model.layers.19.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
151
+ "model.layers.19.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
152
+ "model.layers.19.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
153
+ "model.layers.2.input_layernorm.weight": "model-00002-of-00002.safetensors",
154
+ "model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
155
+ "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
156
+ "model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
157
+ "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
158
+ "model.layers.2.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
159
+ "model.layers.2.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
160
+ "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
161
+ "model.layers.2.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
162
+ "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
163
+ "model.layers.2.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
164
+ "model.layers.2.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
165
+ "model.layers.20.input_layernorm.weight": "model-00001-of-00002.safetensors",
166
+ "model.layers.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
167
+ "model.layers.20.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
168
+ "model.layers.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
169
+ "model.layers.20.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
170
+ "model.layers.20.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
171
+ "model.layers.20.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
172
+ "model.layers.20.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
173
+ "model.layers.20.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
174
+ "model.layers.20.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
175
+ "model.layers.20.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
176
+ "model.layers.20.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
177
+ "model.layers.21.input_layernorm.weight": "model-00001-of-00002.safetensors",
178
+ "model.layers.21.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
179
+ "model.layers.21.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
180
+ "model.layers.21.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
181
+ "model.layers.21.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
182
+ "model.layers.21.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
183
+ "model.layers.21.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
184
+ "model.layers.21.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
185
+ "model.layers.21.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
186
+ "model.layers.21.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
187
+ "model.layers.21.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
188
+ "model.layers.21.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
189
+ "model.layers.22.input_layernorm.weight": "model-00002-of-00002.safetensors",
190
+ "model.layers.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
191
+ "model.layers.22.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
192
+ "model.layers.22.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
193
+ "model.layers.22.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
194
+ "model.layers.22.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
195
+ "model.layers.22.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
196
+ "model.layers.22.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
197
+ "model.layers.22.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
198
+ "model.layers.22.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
199
+ "model.layers.22.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
200
+ "model.layers.22.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
201
+ "model.layers.23.input_layernorm.weight": "model-00001-of-00002.safetensors",
202
+ "model.layers.23.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
203
+ "model.layers.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
204
+ "model.layers.23.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
205
+ "model.layers.23.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
206
+ "model.layers.23.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
207
+ "model.layers.23.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
208
+ "model.layers.23.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
209
+ "model.layers.23.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
210
+ "model.layers.23.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
211
+ "model.layers.23.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
212
+ "model.layers.23.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
213
+ "model.layers.24.input_layernorm.weight": "model-00001-of-00002.safetensors",
214
+ "model.layers.24.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
215
+ "model.layers.24.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
216
+ "model.layers.24.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
217
+ "model.layers.24.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
218
+ "model.layers.24.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
219
+ "model.layers.24.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
220
+ "model.layers.24.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
221
+ "model.layers.24.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
222
+ "model.layers.24.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
223
+ "model.layers.24.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
224
+ "model.layers.24.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
225
+ "model.layers.25.input_layernorm.weight": "model-00001-of-00002.safetensors",
226
+ "model.layers.25.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
227
+ "model.layers.25.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
228
+ "model.layers.25.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
229
+ "model.layers.25.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
230
+ "model.layers.25.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
231
+ "model.layers.25.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
232
+ "model.layers.25.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
233
+ "model.layers.25.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
234
+ "model.layers.25.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
235
+ "model.layers.25.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
236
+ "model.layers.25.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
237
+ "model.layers.26.input_layernorm.weight": "model-00001-of-00002.safetensors",
238
+ "model.layers.26.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
239
+ "model.layers.26.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
240
+ "model.layers.26.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
241
+ "model.layers.26.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
242
+ "model.layers.26.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
243
+ "model.layers.26.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
244
+ "model.layers.26.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
245
+ "model.layers.26.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
246
+ "model.layers.26.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
247
+ "model.layers.26.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
248
+ "model.layers.26.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
249
+ "model.layers.27.input_layernorm.weight": "model-00002-of-00002.safetensors",
250
+ "model.layers.27.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
251
+ "model.layers.27.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
252
+ "model.layers.27.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
253
+ "model.layers.27.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
254
+ "model.layers.27.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
255
+ "model.layers.27.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
256
+ "model.layers.27.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
257
+ "model.layers.27.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
258
+ "model.layers.27.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
259
+ "model.layers.27.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
260
+ "model.layers.27.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
261
+ "model.layers.28.input_layernorm.weight": "model-00001-of-00002.safetensors",
262
+ "model.layers.28.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
263
+ "model.layers.28.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
264
+ "model.layers.28.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
265
+ "model.layers.28.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
266
+ "model.layers.28.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
267
+ "model.layers.28.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
268
+ "model.layers.28.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
269
+ "model.layers.28.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
270
+ "model.layers.28.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
271
+ "model.layers.28.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
272
+ "model.layers.28.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
273
+ "model.layers.29.input_layernorm.weight": "model-00001-of-00002.safetensors",
274
+ "model.layers.29.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
275
+ "model.layers.29.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
276
+ "model.layers.29.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
277
+ "model.layers.29.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
278
+ "model.layers.29.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
279
+ "model.layers.29.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
280
+ "model.layers.29.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
281
+ "model.layers.29.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
282
+ "model.layers.29.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
283
+ "model.layers.29.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
284
+ "model.layers.29.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
285
+ "model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors",
286
+ "model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
287
+ "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
288
+ "model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
289
+ "model.layers.3.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
290
+ "model.layers.3.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
291
+ "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
292
+ "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
293
+ "model.layers.3.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
294
+ "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
295
+ "model.layers.3.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
296
+ "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
297
+ "model.layers.30.input_layernorm.weight": "model-00001-of-00002.safetensors",
298
+ "model.layers.30.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
299
+ "model.layers.30.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
300
+ "model.layers.30.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
301
+ "model.layers.30.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
302
+ "model.layers.30.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
303
+ "model.layers.30.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
304
+ "model.layers.30.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
305
+ "model.layers.30.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
306
+ "model.layers.30.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
307
+ "model.layers.30.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
308
+ "model.layers.30.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
309
+ "model.layers.31.input_layernorm.weight": "model-00001-of-00002.safetensors",
310
+ "model.layers.31.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
311
+ "model.layers.31.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
312
+ "model.layers.31.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
313
+ "model.layers.31.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
314
+ "model.layers.31.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
315
+ "model.layers.31.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
316
+ "model.layers.31.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
317
+ "model.layers.31.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
318
+ "model.layers.31.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
319
+ "model.layers.31.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
320
+ "model.layers.31.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
321
+ "model.layers.32.input_layernorm.weight": "model-00001-of-00002.safetensors",
322
+ "model.layers.32.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
323
+ "model.layers.32.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
324
+ "model.layers.32.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
325
+ "model.layers.32.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
326
+ "model.layers.32.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
327
+ "model.layers.32.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
328
+ "model.layers.32.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
329
+ "model.layers.32.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
330
+ "model.layers.32.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
331
+ "model.layers.32.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
332
+ "model.layers.32.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
333
+ "model.layers.33.input_layernorm.weight": "model-00002-of-00002.safetensors",
334
+ "model.layers.33.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
335
+ "model.layers.33.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
336
+ "model.layers.33.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
337
+ "model.layers.33.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
338
+ "model.layers.33.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
339
+ "model.layers.33.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
340
+ "model.layers.33.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
341
+ "model.layers.33.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
342
+ "model.layers.33.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
343
+ "model.layers.33.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
344
+ "model.layers.33.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
345
+ "model.layers.34.input_layernorm.weight": "model-00002-of-00002.safetensors",
346
+ "model.layers.34.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
347
+ "model.layers.34.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
348
+ "model.layers.34.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
349
+ "model.layers.34.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
350
+ "model.layers.34.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
351
+ "model.layers.34.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
352
+ "model.layers.34.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
353
+ "model.layers.34.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
354
+ "model.layers.34.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
355
+ "model.layers.34.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
356
+ "model.layers.34.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
357
+ "model.layers.35.input_layernorm.weight": "model-00001-of-00002.safetensors",
358
+ "model.layers.35.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
359
+ "model.layers.35.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
360
+ "model.layers.35.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
361
+ "model.layers.35.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
362
+ "model.layers.35.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
363
+ "model.layers.35.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
364
+ "model.layers.35.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
365
+ "model.layers.35.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
366
+ "model.layers.35.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
367
+ "model.layers.35.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
368
+ "model.layers.35.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
369
+ "model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors",
370
+ "model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
371
+ "model.layers.4.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
372
+ "model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
373
+ "model.layers.4.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
374
+ "model.layers.4.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
375
+ "model.layers.4.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
376
+ "model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
377
+ "model.layers.4.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
378
+ "model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
379
+ "model.layers.4.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
380
+ "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
381
+ "model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
382
+ "model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
383
+ "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
384
+ "model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
385
+ "model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
386
+ "model.layers.5.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
387
+ "model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
388
+ "model.layers.5.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
389
+ "model.layers.5.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
390
+ "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
391
+ "model.layers.5.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
392
+ "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
393
+ "model.layers.6.input_layernorm.weight": "model-00002-of-00002.safetensors",
394
+ "model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
395
+ "model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
396
+ "model.layers.6.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
397
+ "model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
398
+ "model.layers.6.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
399
+ "model.layers.6.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
400
+ "model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
401
+ "model.layers.6.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
402
+ "model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
403
+ "model.layers.6.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
404
+ "model.layers.6.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
405
+ "model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
406
+ "model.layers.7.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
407
+ "model.layers.7.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
408
+ "model.layers.7.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
409
+ "model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
410
+ "model.layers.7.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
411
+ "model.layers.7.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
412
+ "model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
413
+ "model.layers.7.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
414
+ "model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
415
+ "model.layers.7.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
416
+ "model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
417
+ "model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
418
+ "model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
419
+ "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
420
+ "model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
421
+ "model.layers.8.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
422
+ "model.layers.8.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
423
+ "model.layers.8.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
424
+ "model.layers.8.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
425
+ "model.layers.8.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
426
+ "model.layers.8.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
427
+ "model.layers.8.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
428
+ "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
429
+ "model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
430
+ "model.layers.9.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
431
+ "model.layers.9.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
432
+ "model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
433
+ "model.layers.9.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
434
+ "model.layers.9.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
435
+ "model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
436
+ "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
437
+ "model.layers.9.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
438
+ "model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
439
+ "model.layers.9.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
440
+ "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
441
+ "model.norm.weight": "model-00002-of-00002.safetensors"
442
+ }
443
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/special_tokens_map.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>",
5
+ "<|object_ref_start|>",
6
+ "<|object_ref_end|>",
7
+ "<|box_start|>",
8
+ "<|box_end|>",
9
+ "<|quad_start|>",
10
+ "<|quad_end|>",
11
+ "<|vision_start|>",
12
+ "<|vision_end|>",
13
+ "<|vision_pad|>",
14
+ "<|image_pad|>",
15
+ "<|video_pad|>"
16
+ ],
17
+ "eos_token": {
18
+ "content": "<|endoftext|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": false,
22
+ "single_word": false
23
+ },
24
+ "pad_token": {
25
+ "content": "<|endoftext|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": false,
29
+ "single_word": false
30
+ }
31
+ }
rlrb_bs128_nobuf_seed2/checkpoints_hf_format/global_step_1400/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
3
+ size 11421896