arch-stanton-1 commited on
Commit
04a7abf
·
verified ·
1 Parent(s): a1661b6

Upload folder using huggingface_hub

Browse files
Files changed (42) hide show
  1. .gitattributes +4 -0
  2. README.md +58 -0
  3. chat_template.jinja +54 -0
  4. checkpoint-868/chat_template.jinja +54 -0
  5. checkpoint-868/config.json +57 -0
  6. checkpoint-868/generation_config.json +13 -0
  7. checkpoint-868/model.safetensors +3 -0
  8. checkpoint-868/optimizer.pt +3 -0
  9. checkpoint-868/rng_state.pth +3 -0
  10. checkpoint-868/scheduler.pt +3 -0
  11. checkpoint-868/tokenizer.json +3 -0
  12. checkpoint-868/tokenizer_config.json +16 -0
  13. checkpoint-868/trainer_state.json +451 -0
  14. checkpoint-868/training_args.bin +3 -0
  15. checkpoint-992/chat_template.jinja +54 -0
  16. checkpoint-992/config.json +57 -0
  17. checkpoint-992/generation_config.json +13 -0
  18. checkpoint-992/model.safetensors +3 -0
  19. checkpoint-992/optimizer.pt +3 -0
  20. checkpoint-992/rng_state.pth +3 -0
  21. checkpoint-992/scheduler.pt +3 -0
  22. checkpoint-992/tokenizer.json +3 -0
  23. checkpoint-992/tokenizer_config.json +16 -0
  24. checkpoint-992/trainer_state.json +512 -0
  25. checkpoint-992/training_args.bin +3 -0
  26. checkpoint-996/chat_template.jinja +54 -0
  27. checkpoint-996/config.json +57 -0
  28. checkpoint-996/generation_config.json +13 -0
  29. checkpoint-996/model.safetensors +3 -0
  30. checkpoint-996/optimizer.pt +3 -0
  31. checkpoint-996/rng_state.pth +3 -0
  32. checkpoint-996/scheduler.pt +3 -0
  33. checkpoint-996/tokenizer.json +3 -0
  34. checkpoint-996/tokenizer_config.json +16 -0
  35. checkpoint-996/trainer_state.json +523 -0
  36. checkpoint-996/training_args.bin +3 -0
  37. config.json +57 -0
  38. generation_config.json +13 -0
  39. model.safetensors +3 -0
  40. tokenizer.json +3 -0
  41. tokenizer_config.json +16 -0
  42. training_args.bin +3 -0
.gitattributes CHANGED
@@ -33,3 +33,7 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ checkpoint-868/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ checkpoint-992/tokenizer.json filter=lfs diff=lfs merge=lfs -text
38
+ checkpoint-996/tokenizer.json filter=lfs diff=lfs merge=lfs -text
39
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: neuphonic/neutts-air
3
+ library_name: transformers
4
+ model_name: neutts_air_urdu_finetune
5
+ tags:
6
+ - generated_from_trainer
7
+ - trl
8
+ - sft
9
+ licence: license
10
+ ---
11
+
12
+ # Model Card for neutts_air_urdu_finetune
13
+
14
+ This model is a fine-tuned version of [neuphonic/neutts-air](https://huggingface.co/neuphonic/neutts-air).
15
+ It has been trained using [TRL](https://github.com/huggingface/trl).
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from transformers import pipeline
21
+
22
+ question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
23
+ generator = pipeline("text-generation", model="None", device="cuda")
24
+ output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
25
+ print(output["generated_text"])
26
+ ```
27
+
28
+ ## Training procedure
29
+
30
+
31
+
32
+
33
+
34
+ This model was trained with SFT.
35
+
36
+ ### Framework versions
37
+
38
+ - TRL: 1.9.0
39
+ - Transformers: 5.14.1
40
+ - Pytorch: 2.5.1+cu121
41
+ - Datasets: 2.20.0
42
+ - Tokenizers: 0.22.2
43
+
44
+ ## Citations
45
+
46
+
47
+
48
+ Cite TRL as:
49
+
50
+ ```bibtex
51
+ @software{vonwerra2020trl,
52
+ title = {{TRL: Transformers Reinforcement Learning}},
53
+ author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
54
+ license = {Apache-2.0},
55
+ url = {https://github.com/huggingface/trl},
56
+ year = {2020}
57
+ }
58
+ ```
chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
checkpoint-868/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
checkpoint-868/config.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": null,
7
+ "dtype": "bfloat16",
8
+ "eos_token_id": 151645,
9
+ "hidden_act": "silu",
10
+ "hidden_size": 896,
11
+ "initializer_range": 0.02,
12
+ "intermediate_size": 4864,
13
+ "layer_types": [
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention"
38
+ ],
39
+ "max_position_embeddings": 32768,
40
+ "max_window_layers": 21,
41
+ "model_type": "qwen2",
42
+ "num_attention_heads": 14,
43
+ "num_hidden_layers": 24,
44
+ "num_key_value_heads": 2,
45
+ "pad_token_id": 151645,
46
+ "rms_norm_eps": 1e-06,
47
+ "rope_parameters": {
48
+ "rope_theta": 1000000.0,
49
+ "rope_type": "default"
50
+ },
51
+ "sliding_window": null,
52
+ "tie_word_embeddings": true,
53
+ "transformers_version": "5.14.1",
54
+ "use_cache": false,
55
+ "use_sliding_window": false,
56
+ "vocab_size": 217653
57
+ }
checkpoint-868/generation_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_sample": true,
3
+ "eos_token_id": [
4
+ 151645,
5
+ 151643
6
+ ],
7
+ "pad_token_id": 151645,
8
+ "repetition_penalty": 1.1,
9
+ "temperature": 0.7,
10
+ "top_k": 20,
11
+ "top_p": 0.8,
12
+ "transformers_version": "5.14.1"
13
+ }
checkpoint-868/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:72987f518ab398cf02ffa544b2cda2cbe222546c15f29d63c05ddd8ffc3b6103
3
+ size 1105862784
checkpoint-868/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8a4312747c7bbac8faf7745c48dee104d12b3c8ca20eb114e24b2be24e0445fd
3
+ size 2211903930
checkpoint-868/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5edb34d031c0c2b447f3eaadb401a4c1e7e7e6d8c096e28b7092e01a8bd48c92
3
+ size 14244
checkpoint-868/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1320a1bf51bd881c32f1eb4368f15fe02d9ce12d687b72e0b16c30523ed4a53d
3
+ size 1064
checkpoint-868/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad19af06e958a8bdd6b45bfbba8a683aa333fc56623bd35f3a3c2b1d9c2a103e
3
+ size 24140418
checkpoint-868/tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "is_local": false,
9
+ "local_files_only": false,
10
+ "model_max_length": 2048,
11
+ "pad_token": "<|im_end|>",
12
+ "padding_side": "right",
13
+ "split_special_tokens": false,
14
+ "tokenizer_class": "Qwen2Tokenizer",
15
+ "unk_token": null
16
+ }
checkpoint-868/trainer_state.json ADDED
@@ -0,0 +1,451 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 868,
3
+ "best_metric": 7.3567795753479,
4
+ "best_model_checkpoint": "/workspace/neutts_air_urdu_finetune/checkpoint-868",
5
+ "epoch": 3.6779661016949152,
6
+ "eval_steps": 124,
7
+ "global_step": 868,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 7.697032804489136,
14
+ "epoch": 0.1059322033898305,
15
+ "grad_norm": 4.78125,
16
+ "learning_rate": 1.6000000000000003e-05,
17
+ "loss": 7.989871215820313,
18
+ "mean_token_accuracy": 0.016311284080147745,
19
+ "num_tokens": 218322.0,
20
+ "step": 25
21
+ },
22
+ {
23
+ "entropy": 7.771375341415405,
24
+ "epoch": 0.211864406779661,
25
+ "grad_norm": 2.671875,
26
+ "learning_rate": 1.9980915336317713e-05,
27
+ "loss": 7.803304443359375,
28
+ "mean_token_accuracy": 0.018622982166707516,
29
+ "num_tokens": 437349.0,
30
+ "step": 50
31
+ },
32
+ {
33
+ "entropy": 7.646108131408692,
34
+ "epoch": 0.3177966101694915,
35
+ "grad_norm": 2.421875,
36
+ "learning_rate": 1.989779323028575e-05,
37
+ "loss": 7.629185180664063,
38
+ "mean_token_accuracy": 0.021061680242419244,
39
+ "num_tokens": 658790.0,
40
+ "step": 75
41
+ },
42
+ {
43
+ "entropy": 7.555726566314697,
44
+ "epoch": 0.423728813559322,
45
+ "grad_norm": 2.359375,
46
+ "learning_rate": 1.9749279121818235e-05,
47
+ "loss": 7.55064208984375,
48
+ "mean_token_accuracy": 0.022715183552354575,
49
+ "num_tokens": 872825.0,
50
+ "step": 100
51
+ },
52
+ {
53
+ "epoch": 0.5254237288135594,
54
+ "eval_entropy": 7.486619492371877,
55
+ "eval_loss": 7.498717784881592,
56
+ "eval_mean_token_accuracy": 0.02434113745888074,
57
+ "eval_num_tokens": 1084707.0,
58
+ "eval_runtime": 7.3619,
59
+ "eval_samples_per_second": 54.062,
60
+ "eval_steps_per_second": 3.396,
61
+ "step": 124
62
+ },
63
+ {
64
+ "entropy": 7.509912757873535,
65
+ "epoch": 0.5296610169491526,
66
+ "grad_norm": 2.21875,
67
+ "learning_rate": 1.9536354202855306e-05,
68
+ "loss": 7.5065460205078125,
69
+ "mean_token_accuracy": 0.022764644995331765,
70
+ "num_tokens": 1093864.0,
71
+ "step": 125
72
+ },
73
+ {
74
+ "entropy": 7.504309902191162,
75
+ "epoch": 0.635593220338983,
76
+ "grad_norm": 2.078125,
77
+ "learning_rate": 1.9260425209878052e-05,
78
+ "loss": 7.493028564453125,
79
+ "mean_token_accuracy": 0.022260171882808207,
80
+ "num_tokens": 1315198.0,
81
+ "step": 150
82
+ },
83
+ {
84
+ "entropy": 7.464480457305908,
85
+ "epoch": 0.7415254237288136,
86
+ "grad_norm": 2.15625,
87
+ "learning_rate": 1.8923315129986838e-05,
88
+ "loss": 7.463182373046875,
89
+ "mean_token_accuracy": 0.023391987532377242,
90
+ "num_tokens": 1533213.0,
91
+ "step": 175
92
+ },
93
+ {
94
+ "entropy": 7.452612771987915,
95
+ "epoch": 0.847457627118644,
96
+ "grad_norm": 2.34375,
97
+ "learning_rate": 1.8527251156925997e-05,
98
+ "loss": 7.447894897460937,
99
+ "mean_token_accuracy": 0.02403868570923805,
100
+ "num_tokens": 1752307.0,
101
+ "step": 200
102
+ },
103
+ {
104
+ "entropy": 7.409109563827514,
105
+ "epoch": 0.9533898305084746,
106
+ "grad_norm": 1.96875,
107
+ "learning_rate": 1.8074849976626273e-05,
108
+ "loss": 7.407470703125,
109
+ "mean_token_accuracy": 0.023314249143004416,
110
+ "num_tokens": 1973350.0,
111
+ "step": 225
112
+ },
113
+ {
114
+ "epoch": 1.0508474576271187,
115
+ "eval_entropy": 7.401865323384603,
116
+ "eval_loss": 7.414607524871826,
117
+ "eval_mean_token_accuracy": 0.025272298449029524,
118
+ "eval_num_tokens": 2178943.0,
119
+ "eval_runtime": 7.4146,
120
+ "eval_samples_per_second": 53.678,
121
+ "eval_steps_per_second": 3.372,
122
+ "step": 248
123
+ },
124
+ {
125
+ "entropy": 7.410350885391235,
126
+ "epoch": 1.0593220338983051,
127
+ "grad_norm": 2.078125,
128
+ "learning_rate": 1.756910047947926e-05,
129
+ "loss": 7.405702514648437,
130
+ "mean_token_accuracy": 0.024052796624600886,
131
+ "num_tokens": 2195473.0,
132
+ "step": 250
133
+ },
134
+ {
135
+ "entropy": 7.378520240783692,
136
+ "epoch": 1.1652542372881356,
137
+ "grad_norm": 2.0625,
138
+ "learning_rate": 1.7013344013559197e-05,
139
+ "loss": 7.368658447265625,
140
+ "mean_token_accuracy": 0.024010552018880846,
141
+ "num_tokens": 2410539.0,
142
+ "step": 275
143
+ },
144
+ {
145
+ "entropy": 7.370533514022827,
146
+ "epoch": 1.271186440677966,
147
+ "grad_norm": 2.015625,
148
+ "learning_rate": 1.64112523092536e-05,
149
+ "loss": 7.370477905273438,
150
+ "mean_token_accuracy": 0.02511810462921858,
151
+ "num_tokens": 2630347.0,
152
+ "step": 300
153
+ },
154
+ {
155
+ "entropy": 7.372100067138672,
156
+ "epoch": 1.3771186440677967,
157
+ "grad_norm": 2.140625,
158
+ "learning_rate": 1.5766803221148676e-05,
159
+ "loss": 7.374469604492187,
160
+ "mean_token_accuracy": 0.02450455617159605,
161
+ "num_tokens": 2848534.0,
162
+ "step": 325
163
+ },
164
+ {
165
+ "entropy": 7.342970514297486,
166
+ "epoch": 1.4830508474576272,
167
+ "grad_norm": 1.9609375,
168
+ "learning_rate": 1.5084254447436169e-05,
169
+ "loss": 7.339466552734375,
170
+ "mean_token_accuracy": 0.025138991102576256,
171
+ "num_tokens": 3070123.0,
172
+ "step": 350
173
+ },
174
+ {
175
+ "epoch": 1.576271186440678,
176
+ "eval_entropy": 7.361576656500499,
177
+ "eval_loss": 7.381193161010742,
178
+ "eval_mean_token_accuracy": 0.025545974417279165,
179
+ "eval_num_tokens": 3263577.0,
180
+ "eval_runtime": 7.4123,
181
+ "eval_samples_per_second": 53.695,
182
+ "eval_steps_per_second": 3.373,
183
+ "step": 372
184
+ },
185
+ {
186
+ "entropy": 7.361075611114502,
187
+ "epoch": 1.5889830508474576,
188
+ "grad_norm": 1.9140625,
189
+ "learning_rate": 1.4368115400470393e-05,
190
+ "loss": 7.3520880126953125,
191
+ "mean_token_accuracy": 0.024180141538381578,
192
+ "num_tokens": 3288806.0,
193
+ "step": 375
194
+ },
195
+ {
196
+ "entropy": 7.3546325302124025,
197
+ "epoch": 1.694915254237288,
198
+ "grad_norm": 2.15625,
199
+ "learning_rate": 1.3623117414318827e-05,
200
+ "loss": 7.346293334960937,
201
+ "mean_token_accuracy": 0.024999124109745027,
202
+ "num_tokens": 3510776.0,
203
+ "step": 400
204
+ },
205
+ {
206
+ "entropy": 7.363360710144043,
207
+ "epoch": 1.8008474576271185,
208
+ "grad_norm": 2.078125,
209
+ "learning_rate": 1.2854182486136944e-05,
210
+ "loss": 7.364286499023438,
211
+ "mean_token_accuracy": 0.02418241061270237,
212
+ "num_tokens": 3731405.0,
213
+ "step": 425
214
+ },
215
+ {
216
+ "entropy": 7.334566602706909,
217
+ "epoch": 1.9067796610169492,
218
+ "grad_norm": 1.9609375,
219
+ "learning_rate": 1.2066390757884328e-05,
220
+ "loss": 7.33093505859375,
221
+ "mean_token_accuracy": 0.02495731335133314,
222
+ "num_tokens": 3953207.0,
223
+ "step": 450
224
+ },
225
+ {
226
+ "entropy": 7.353079776763916,
227
+ "epoch": 2.01271186440678,
228
+ "grad_norm": 1.8671875,
229
+ "learning_rate": 1.1264946953221496e-05,
230
+ "loss": 7.351827392578125,
231
+ "mean_token_accuracy": 0.02447190221399069,
232
+ "num_tokens": 4169521.0,
233
+ "step": 475
234
+ },
235
+ {
236
+ "epoch": 2.1016949152542375,
237
+ "eval_entropy": 7.34862866004308,
238
+ "eval_loss": 7.365415096282959,
239
+ "eval_mean_token_accuracy": 0.025626841001212597,
240
+ "eval_num_tokens": 4351452.0,
241
+ "eval_runtime": 7.4142,
242
+ "eval_samples_per_second": 53.681,
243
+ "eval_steps_per_second": 3.372,
244
+ "step": 496
245
+ },
246
+ {
247
+ "entropy": 7.339797382354736,
248
+ "epoch": 2.1186440677966103,
249
+ "grad_norm": 2.03125,
250
+ "learning_rate": 1.0455145991329639e-05,
251
+ "loss": 7.3259820556640625,
252
+ "mean_token_accuracy": 0.025096801929175853,
253
+ "num_tokens": 4386912.0,
254
+ "step": 500
255
+ },
256
+ {
257
+ "entropy": 7.330105533599854,
258
+ "epoch": 2.2245762711864407,
259
+ "grad_norm": 2.03125,
260
+ "learning_rate": 9.642338004833295e-06,
261
+ "loss": 7.330958251953125,
262
+ "mean_token_accuracy": 0.024473249688744547,
263
+ "num_tokens": 4607819.0,
264
+ "step": 525
265
+ },
266
+ {
267
+ "entropy": 7.331509189605713,
268
+ "epoch": 2.330508474576271,
269
+ "grad_norm": 1.96875,
270
+ "learning_rate": 8.831892992943e-06,
271
+ "loss": 7.329585571289062,
272
+ "mean_token_accuracy": 0.024803164564073086,
273
+ "num_tokens": 4829926.0,
274
+ "step": 550
275
+ },
276
+ {
277
+ "entropy": 7.332346143722535,
278
+ "epoch": 2.4364406779661016,
279
+ "grad_norm": 1.953125,
280
+ "learning_rate": 8.029165343344805e-06,
281
+ "loss": 7.3258428955078125,
282
+ "mean_token_accuracy": 0.02469826426357031,
283
+ "num_tokens": 5045115.0,
284
+ "step": 575
285
+ },
286
+ {
287
+ "entropy": 7.318505411148071,
288
+ "epoch": 2.542372881355932,
289
+ "grad_norm": 1.9140625,
290
+ "learning_rate": 7.2394584572309125e-06,
291
+ "loss": 7.3118963623046875,
292
+ "mean_token_accuracy": 0.02518722042441368,
293
+ "num_tokens": 5265448.0,
294
+ "step": 600
295
+ },
296
+ {
297
+ "epoch": 2.6271186440677967,
298
+ "eval_entropy": 7.333844860394795,
299
+ "eval_loss": 7.359093189239502,
300
+ "eval_mean_token_accuracy": 0.025550531456246972,
301
+ "eval_num_tokens": 5439386.0,
302
+ "eval_runtime": 7.4052,
303
+ "eval_samples_per_second": 53.746,
304
+ "eval_steps_per_second": 3.376,
305
+ "step": 620
306
+ },
307
+ {
308
+ "entropy": 7.321087207794189,
309
+ "epoch": 2.648305084745763,
310
+ "grad_norm": 2.0625,
311
+ "learning_rate": 6.467989711184021e-06,
312
+ "loss": 7.31301513671875,
313
+ "mean_token_accuracy": 0.025983325839042663,
314
+ "num_tokens": 5482207.0,
315
+ "step": 625
316
+ },
317
+ {
318
+ "entropy": 7.322469959259033,
319
+ "epoch": 2.7542372881355934,
320
+ "grad_norm": 2.125,
321
+ "learning_rate": 5.7198559874026945e-06,
322
+ "loss": 7.318033447265625,
323
+ "mean_token_accuracy": 0.02565582599490881,
324
+ "num_tokens": 5703727.0,
325
+ "step": 650
326
+ },
327
+ {
328
+ "entropy": 7.31039176940918,
329
+ "epoch": 2.860169491525424,
330
+ "grad_norm": 2.03125,
331
+ "learning_rate": 5.000000000000003e-06,
332
+ "loss": 7.305034790039063,
333
+ "mean_token_accuracy": 0.02535436548292637,
334
+ "num_tokens": 5923157.0,
335
+ "step": 675
336
+ },
337
+ {
338
+ "entropy": 7.3113681411743165,
339
+ "epoch": 2.9661016949152543,
340
+ "grad_norm": 1.9765625,
341
+ "learning_rate": 4.313177639848408e-06,
342
+ "loss": 7.3114166259765625,
343
+ "mean_token_accuracy": 0.025559407025575638,
344
+ "num_tokens": 6143779.0,
345
+ "step": 700
346
+ },
347
+ {
348
+ "entropy": 7.3387535953521725,
349
+ "epoch": 3.0720338983050848,
350
+ "grad_norm": 1.921875,
351
+ "learning_rate": 3.6639265537145187e-06,
352
+ "loss": 7.343036499023437,
353
+ "mean_token_accuracy": 0.024094666354358196,
354
+ "num_tokens": 6365371.0,
355
+ "step": 725
356
+ },
357
+ {
358
+ "epoch": 3.152542372881356,
359
+ "eval_entropy": 7.332533776760101,
360
+ "eval_loss": 7.357120990753174,
361
+ "eval_mean_token_accuracy": 0.025878646954273183,
362
+ "eval_num_tokens": 6532337.0,
363
+ "eval_runtime": 7.4033,
364
+ "eval_samples_per_second": 53.76,
365
+ "eval_steps_per_second": 3.377,
366
+ "step": 744
367
+ },
368
+ {
369
+ "entropy": 7.320968351364136,
370
+ "epoch": 3.1779661016949152,
371
+ "grad_norm": 2.109375,
372
+ "learning_rate": 3.056536165272662e-06,
373
+ "loss": 7.313834228515625,
374
+ "mean_token_accuracy": 0.025647504702210427,
375
+ "num_tokens": 6584143.0,
376
+ "step": 750
377
+ },
378
+ {
379
+ "entropy": 7.319151067733765,
380
+ "epoch": 3.2838983050847457,
381
+ "grad_norm": 1.9140625,
382
+ "learning_rate": 2.4950193360603868e-06,
383
+ "loss": 7.315574951171875,
384
+ "mean_token_accuracy": 0.02516275253146887,
385
+ "num_tokens": 6806162.0,
386
+ "step": 775
387
+ },
388
+ {
389
+ "entropy": 7.312239484786987,
390
+ "epoch": 3.389830508474576,
391
+ "grad_norm": 1.8359375,
392
+ "learning_rate": 1.9830858536039742e-06,
393
+ "loss": 7.299415893554688,
394
+ "mean_token_accuracy": 0.025689680464565753,
395
+ "num_tokens": 7025188.0,
396
+ "step": 800
397
+ },
398
+ {
399
+ "entropy": 7.3349973487854,
400
+ "epoch": 3.4957627118644066,
401
+ "grad_norm": 1.90625,
402
+ "learning_rate": 1.524117921870789e-06,
403
+ "loss": 7.336126098632812,
404
+ "mean_token_accuracy": 0.02420500263571739,
405
+ "num_tokens": 7242603.0,
406
+ "step": 825
407
+ },
408
+ {
409
+ "entropy": 7.306003942489624,
410
+ "epoch": 3.601694915254237,
411
+ "grad_norm": 1.765625,
412
+ "learning_rate": 1.121147815976248e-06,
413
+ "loss": 7.2987548828125,
414
+ "mean_token_accuracy": 0.02604618974030018,
415
+ "num_tokens": 7462100.0,
416
+ "step": 850
417
+ },
418
+ {
419
+ "epoch": 3.6779661016949152,
420
+ "eval_entropy": 7.331447899341583,
421
+ "eval_loss": 7.3567795753479,
422
+ "eval_mean_token_accuracy": 0.02568032095829646,
423
+ "eval_num_tokens": 7621408.0,
424
+ "eval_runtime": 7.4184,
425
+ "eval_samples_per_second": 53.65,
426
+ "eval_steps_per_second": 3.37,
427
+ "step": 868
428
+ }
429
+ ],
430
+ "logging_steps": 25,
431
+ "max_steps": 996,
432
+ "num_input_tokens_seen": 0,
433
+ "num_train_epochs": 5,
434
+ "save_steps": 124,
435
+ "stateful_callbacks": {
436
+ "TrainerControl": {
437
+ "args": {
438
+ "should_epoch_stop": false,
439
+ "should_evaluate": false,
440
+ "should_log": false,
441
+ "should_save": true,
442
+ "should_training_stop": false
443
+ },
444
+ "attributes": {}
445
+ }
446
+ },
447
+ "total_flos": 3.817335536222208e+16,
448
+ "train_batch_size": 16,
449
+ "trial_name": null,
450
+ "trial_params": null
451
+ }
checkpoint-868/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51fb563566aef1daee880df75431bb65187482b5a42194bffca0aab31b9931d2
3
+ size 5368
checkpoint-992/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
checkpoint-992/config.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": null,
7
+ "dtype": "bfloat16",
8
+ "eos_token_id": 151645,
9
+ "hidden_act": "silu",
10
+ "hidden_size": 896,
11
+ "initializer_range": 0.02,
12
+ "intermediate_size": 4864,
13
+ "layer_types": [
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention"
38
+ ],
39
+ "max_position_embeddings": 32768,
40
+ "max_window_layers": 21,
41
+ "model_type": "qwen2",
42
+ "num_attention_heads": 14,
43
+ "num_hidden_layers": 24,
44
+ "num_key_value_heads": 2,
45
+ "pad_token_id": 151645,
46
+ "rms_norm_eps": 1e-06,
47
+ "rope_parameters": {
48
+ "rope_theta": 1000000.0,
49
+ "rope_type": "default"
50
+ },
51
+ "sliding_window": null,
52
+ "tie_word_embeddings": true,
53
+ "transformers_version": "5.14.1",
54
+ "use_cache": false,
55
+ "use_sliding_window": false,
56
+ "vocab_size": 217653
57
+ }
checkpoint-992/generation_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_sample": true,
3
+ "eos_token_id": [
4
+ 151645,
5
+ 151643
6
+ ],
7
+ "pad_token_id": 151645,
8
+ "repetition_penalty": 1.1,
9
+ "temperature": 0.7,
10
+ "top_k": 20,
11
+ "top_p": 0.8,
12
+ "transformers_version": "5.14.1"
13
+ }
checkpoint-992/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6217e5cd2f1fcb6b5775d40c6af19bef77ccfc0785719c2a3fea3c27f9d046f8
3
+ size 1105862784
checkpoint-992/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a0587cf0b401c232cd599d6a5c22c709a7ab4d53cc68d6f142a05fa70b3f29c7
3
+ size 2211903930
checkpoint-992/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7682299c566684ea51cf26f0c86b6ffaa3c0bc63cbdf84674b29a2c62ac72143
3
+ size 14244
checkpoint-992/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:582bcaa1a2856315614a378240a5cabaac9c30c01b2ac43af3afa892908adea3
3
+ size 1064
checkpoint-992/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad19af06e958a8bdd6b45bfbba8a683aa333fc56623bd35f3a3c2b1d9c2a103e
3
+ size 24140418
checkpoint-992/tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "is_local": false,
9
+ "local_files_only": false,
10
+ "model_max_length": 2048,
11
+ "pad_token": "<|im_end|>",
12
+ "padding_side": "right",
13
+ "split_special_tokens": false,
14
+ "tokenizer_class": "Qwen2Tokenizer",
15
+ "unk_token": null
16
+ }
checkpoint-992/trainer_state.json ADDED
@@ -0,0 +1,512 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 992,
3
+ "best_metric": 7.356758117675781,
4
+ "best_model_checkpoint": "/workspace/neutts_air_urdu_finetune/checkpoint-992",
5
+ "epoch": 4.203389830508475,
6
+ "eval_steps": 124,
7
+ "global_step": 992,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 7.697032804489136,
14
+ "epoch": 0.1059322033898305,
15
+ "grad_norm": 4.78125,
16
+ "learning_rate": 1.6000000000000003e-05,
17
+ "loss": 7.989871215820313,
18
+ "mean_token_accuracy": 0.016311284080147745,
19
+ "num_tokens": 218322.0,
20
+ "step": 25
21
+ },
22
+ {
23
+ "entropy": 7.771375341415405,
24
+ "epoch": 0.211864406779661,
25
+ "grad_norm": 2.671875,
26
+ "learning_rate": 1.9980915336317713e-05,
27
+ "loss": 7.803304443359375,
28
+ "mean_token_accuracy": 0.018622982166707516,
29
+ "num_tokens": 437349.0,
30
+ "step": 50
31
+ },
32
+ {
33
+ "entropy": 7.646108131408692,
34
+ "epoch": 0.3177966101694915,
35
+ "grad_norm": 2.421875,
36
+ "learning_rate": 1.989779323028575e-05,
37
+ "loss": 7.629185180664063,
38
+ "mean_token_accuracy": 0.021061680242419244,
39
+ "num_tokens": 658790.0,
40
+ "step": 75
41
+ },
42
+ {
43
+ "entropy": 7.555726566314697,
44
+ "epoch": 0.423728813559322,
45
+ "grad_norm": 2.359375,
46
+ "learning_rate": 1.9749279121818235e-05,
47
+ "loss": 7.55064208984375,
48
+ "mean_token_accuracy": 0.022715183552354575,
49
+ "num_tokens": 872825.0,
50
+ "step": 100
51
+ },
52
+ {
53
+ "epoch": 0.5254237288135594,
54
+ "eval_entropy": 7.486619492371877,
55
+ "eval_loss": 7.498717784881592,
56
+ "eval_mean_token_accuracy": 0.02434113745888074,
57
+ "eval_num_tokens": 1084707.0,
58
+ "eval_runtime": 7.3619,
59
+ "eval_samples_per_second": 54.062,
60
+ "eval_steps_per_second": 3.396,
61
+ "step": 124
62
+ },
63
+ {
64
+ "entropy": 7.509912757873535,
65
+ "epoch": 0.5296610169491526,
66
+ "grad_norm": 2.21875,
67
+ "learning_rate": 1.9536354202855306e-05,
68
+ "loss": 7.5065460205078125,
69
+ "mean_token_accuracy": 0.022764644995331765,
70
+ "num_tokens": 1093864.0,
71
+ "step": 125
72
+ },
73
+ {
74
+ "entropy": 7.504309902191162,
75
+ "epoch": 0.635593220338983,
76
+ "grad_norm": 2.078125,
77
+ "learning_rate": 1.9260425209878052e-05,
78
+ "loss": 7.493028564453125,
79
+ "mean_token_accuracy": 0.022260171882808207,
80
+ "num_tokens": 1315198.0,
81
+ "step": 150
82
+ },
83
+ {
84
+ "entropy": 7.464480457305908,
85
+ "epoch": 0.7415254237288136,
86
+ "grad_norm": 2.15625,
87
+ "learning_rate": 1.8923315129986838e-05,
88
+ "loss": 7.463182373046875,
89
+ "mean_token_accuracy": 0.023391987532377242,
90
+ "num_tokens": 1533213.0,
91
+ "step": 175
92
+ },
93
+ {
94
+ "entropy": 7.452612771987915,
95
+ "epoch": 0.847457627118644,
96
+ "grad_norm": 2.34375,
97
+ "learning_rate": 1.8527251156925997e-05,
98
+ "loss": 7.447894897460937,
99
+ "mean_token_accuracy": 0.02403868570923805,
100
+ "num_tokens": 1752307.0,
101
+ "step": 200
102
+ },
103
+ {
104
+ "entropy": 7.409109563827514,
105
+ "epoch": 0.9533898305084746,
106
+ "grad_norm": 1.96875,
107
+ "learning_rate": 1.8074849976626273e-05,
108
+ "loss": 7.407470703125,
109
+ "mean_token_accuracy": 0.023314249143004416,
110
+ "num_tokens": 1973350.0,
111
+ "step": 225
112
+ },
113
+ {
114
+ "epoch": 1.0508474576271187,
115
+ "eval_entropy": 7.401865323384603,
116
+ "eval_loss": 7.414607524871826,
117
+ "eval_mean_token_accuracy": 0.025272298449029524,
118
+ "eval_num_tokens": 2178943.0,
119
+ "eval_runtime": 7.4146,
120
+ "eval_samples_per_second": 53.678,
121
+ "eval_steps_per_second": 3.372,
122
+ "step": 248
123
+ },
124
+ {
125
+ "entropy": 7.410350885391235,
126
+ "epoch": 1.0593220338983051,
127
+ "grad_norm": 2.078125,
128
+ "learning_rate": 1.756910047947926e-05,
129
+ "loss": 7.405702514648437,
130
+ "mean_token_accuracy": 0.024052796624600886,
131
+ "num_tokens": 2195473.0,
132
+ "step": 250
133
+ },
134
+ {
135
+ "entropy": 7.378520240783692,
136
+ "epoch": 1.1652542372881356,
137
+ "grad_norm": 2.0625,
138
+ "learning_rate": 1.7013344013559197e-05,
139
+ "loss": 7.368658447265625,
140
+ "mean_token_accuracy": 0.024010552018880846,
141
+ "num_tokens": 2410539.0,
142
+ "step": 275
143
+ },
144
+ {
145
+ "entropy": 7.370533514022827,
146
+ "epoch": 1.271186440677966,
147
+ "grad_norm": 2.015625,
148
+ "learning_rate": 1.64112523092536e-05,
149
+ "loss": 7.370477905273438,
150
+ "mean_token_accuracy": 0.02511810462921858,
151
+ "num_tokens": 2630347.0,
152
+ "step": 300
153
+ },
154
+ {
155
+ "entropy": 7.372100067138672,
156
+ "epoch": 1.3771186440677967,
157
+ "grad_norm": 2.140625,
158
+ "learning_rate": 1.5766803221148676e-05,
159
+ "loss": 7.374469604492187,
160
+ "mean_token_accuracy": 0.02450455617159605,
161
+ "num_tokens": 2848534.0,
162
+ "step": 325
163
+ },
164
+ {
165
+ "entropy": 7.342970514297486,
166
+ "epoch": 1.4830508474576272,
167
+ "grad_norm": 1.9609375,
168
+ "learning_rate": 1.5084254447436169e-05,
169
+ "loss": 7.339466552734375,
170
+ "mean_token_accuracy": 0.025138991102576256,
171
+ "num_tokens": 3070123.0,
172
+ "step": 350
173
+ },
174
+ {
175
+ "epoch": 1.576271186440678,
176
+ "eval_entropy": 7.361576656500499,
177
+ "eval_loss": 7.381193161010742,
178
+ "eval_mean_token_accuracy": 0.025545974417279165,
179
+ "eval_num_tokens": 3263577.0,
180
+ "eval_runtime": 7.4123,
181
+ "eval_samples_per_second": 53.695,
182
+ "eval_steps_per_second": 3.373,
183
+ "step": 372
184
+ },
185
+ {
186
+ "entropy": 7.361075611114502,
187
+ "epoch": 1.5889830508474576,
188
+ "grad_norm": 1.9140625,
189
+ "learning_rate": 1.4368115400470393e-05,
190
+ "loss": 7.3520880126953125,
191
+ "mean_token_accuracy": 0.024180141538381578,
192
+ "num_tokens": 3288806.0,
193
+ "step": 375
194
+ },
195
+ {
196
+ "entropy": 7.3546325302124025,
197
+ "epoch": 1.694915254237288,
198
+ "grad_norm": 2.15625,
199
+ "learning_rate": 1.3623117414318827e-05,
200
+ "loss": 7.346293334960937,
201
+ "mean_token_accuracy": 0.024999124109745027,
202
+ "num_tokens": 3510776.0,
203
+ "step": 400
204
+ },
205
+ {
206
+ "entropy": 7.363360710144043,
207
+ "epoch": 1.8008474576271185,
208
+ "grad_norm": 2.078125,
209
+ "learning_rate": 1.2854182486136944e-05,
210
+ "loss": 7.364286499023438,
211
+ "mean_token_accuracy": 0.02418241061270237,
212
+ "num_tokens": 3731405.0,
213
+ "step": 425
214
+ },
215
+ {
216
+ "entropy": 7.334566602706909,
217
+ "epoch": 1.9067796610169492,
218
+ "grad_norm": 1.9609375,
219
+ "learning_rate": 1.2066390757884328e-05,
220
+ "loss": 7.33093505859375,
221
+ "mean_token_accuracy": 0.02495731335133314,
222
+ "num_tokens": 3953207.0,
223
+ "step": 450
224
+ },
225
+ {
226
+ "entropy": 7.353079776763916,
227
+ "epoch": 2.01271186440678,
228
+ "grad_norm": 1.8671875,
229
+ "learning_rate": 1.1264946953221496e-05,
230
+ "loss": 7.351827392578125,
231
+ "mean_token_accuracy": 0.02447190221399069,
232
+ "num_tokens": 4169521.0,
233
+ "step": 475
234
+ },
235
+ {
236
+ "epoch": 2.1016949152542375,
237
+ "eval_entropy": 7.34862866004308,
238
+ "eval_loss": 7.365415096282959,
239
+ "eval_mean_token_accuracy": 0.025626841001212597,
240
+ "eval_num_tokens": 4351452.0,
241
+ "eval_runtime": 7.4142,
242
+ "eval_samples_per_second": 53.681,
243
+ "eval_steps_per_second": 3.372,
244
+ "step": 496
245
+ },
246
+ {
247
+ "entropy": 7.339797382354736,
248
+ "epoch": 2.1186440677966103,
249
+ "grad_norm": 2.03125,
250
+ "learning_rate": 1.0455145991329639e-05,
251
+ "loss": 7.3259820556640625,
252
+ "mean_token_accuracy": 0.025096801929175853,
253
+ "num_tokens": 4386912.0,
254
+ "step": 500
255
+ },
256
+ {
257
+ "entropy": 7.330105533599854,
258
+ "epoch": 2.2245762711864407,
259
+ "grad_norm": 2.03125,
260
+ "learning_rate": 9.642338004833295e-06,
261
+ "loss": 7.330958251953125,
262
+ "mean_token_accuracy": 0.024473249688744547,
263
+ "num_tokens": 4607819.0,
264
+ "step": 525
265
+ },
266
+ {
267
+ "entropy": 7.331509189605713,
268
+ "epoch": 2.330508474576271,
269
+ "grad_norm": 1.96875,
270
+ "learning_rate": 8.831892992943e-06,
271
+ "loss": 7.329585571289062,
272
+ "mean_token_accuracy": 0.024803164564073086,
273
+ "num_tokens": 4829926.0,
274
+ "step": 550
275
+ },
276
+ {
277
+ "entropy": 7.332346143722535,
278
+ "epoch": 2.4364406779661016,
279
+ "grad_norm": 1.953125,
280
+ "learning_rate": 8.029165343344805e-06,
281
+ "loss": 7.3258428955078125,
282
+ "mean_token_accuracy": 0.02469826426357031,
283
+ "num_tokens": 5045115.0,
284
+ "step": 575
285
+ },
286
+ {
287
+ "entropy": 7.318505411148071,
288
+ "epoch": 2.542372881355932,
289
+ "grad_norm": 1.9140625,
290
+ "learning_rate": 7.2394584572309125e-06,
291
+ "loss": 7.3118963623046875,
292
+ "mean_token_accuracy": 0.02518722042441368,
293
+ "num_tokens": 5265448.0,
294
+ "step": 600
295
+ },
296
+ {
297
+ "epoch": 2.6271186440677967,
298
+ "eval_entropy": 7.333844860394795,
299
+ "eval_loss": 7.359093189239502,
300
+ "eval_mean_token_accuracy": 0.025550531456246972,
301
+ "eval_num_tokens": 5439386.0,
302
+ "eval_runtime": 7.4052,
303
+ "eval_samples_per_second": 53.746,
304
+ "eval_steps_per_second": 3.376,
305
+ "step": 620
306
+ },
307
+ {
308
+ "entropy": 7.321087207794189,
309
+ "epoch": 2.648305084745763,
310
+ "grad_norm": 2.0625,
311
+ "learning_rate": 6.467989711184021e-06,
312
+ "loss": 7.31301513671875,
313
+ "mean_token_accuracy": 0.025983325839042663,
314
+ "num_tokens": 5482207.0,
315
+ "step": 625
316
+ },
317
+ {
318
+ "entropy": 7.322469959259033,
319
+ "epoch": 2.7542372881355934,
320
+ "grad_norm": 2.125,
321
+ "learning_rate": 5.7198559874026945e-06,
322
+ "loss": 7.318033447265625,
323
+ "mean_token_accuracy": 0.02565582599490881,
324
+ "num_tokens": 5703727.0,
325
+ "step": 650
326
+ },
327
+ {
328
+ "entropy": 7.31039176940918,
329
+ "epoch": 2.860169491525424,
330
+ "grad_norm": 2.03125,
331
+ "learning_rate": 5.000000000000003e-06,
332
+ "loss": 7.305034790039063,
333
+ "mean_token_accuracy": 0.02535436548292637,
334
+ "num_tokens": 5923157.0,
335
+ "step": 675
336
+ },
337
+ {
338
+ "entropy": 7.3113681411743165,
339
+ "epoch": 2.9661016949152543,
340
+ "grad_norm": 1.9765625,
341
+ "learning_rate": 4.313177639848408e-06,
342
+ "loss": 7.3114166259765625,
343
+ "mean_token_accuracy": 0.025559407025575638,
344
+ "num_tokens": 6143779.0,
345
+ "step": 700
346
+ },
347
+ {
348
+ "entropy": 7.3387535953521725,
349
+ "epoch": 3.0720338983050848,
350
+ "grad_norm": 1.921875,
351
+ "learning_rate": 3.6639265537145187e-06,
352
+ "loss": 7.343036499023437,
353
+ "mean_token_accuracy": 0.024094666354358196,
354
+ "num_tokens": 6365371.0,
355
+ "step": 725
356
+ },
357
+ {
358
+ "epoch": 3.152542372881356,
359
+ "eval_entropy": 7.332533776760101,
360
+ "eval_loss": 7.357120990753174,
361
+ "eval_mean_token_accuracy": 0.025878646954273183,
362
+ "eval_num_tokens": 6532337.0,
363
+ "eval_runtime": 7.4033,
364
+ "eval_samples_per_second": 53.76,
365
+ "eval_steps_per_second": 3.377,
366
+ "step": 744
367
+ },
368
+ {
369
+ "entropy": 7.320968351364136,
370
+ "epoch": 3.1779661016949152,
371
+ "grad_norm": 2.109375,
372
+ "learning_rate": 3.056536165272662e-06,
373
+ "loss": 7.313834228515625,
374
+ "mean_token_accuracy": 0.025647504702210427,
375
+ "num_tokens": 6584143.0,
376
+ "step": 750
377
+ },
378
+ {
379
+ "entropy": 7.319151067733765,
380
+ "epoch": 3.2838983050847457,
381
+ "grad_norm": 1.9140625,
382
+ "learning_rate": 2.4950193360603868e-06,
383
+ "loss": 7.315574951171875,
384
+ "mean_token_accuracy": 0.02516275253146887,
385
+ "num_tokens": 6806162.0,
386
+ "step": 775
387
+ },
388
+ {
389
+ "entropy": 7.312239484786987,
390
+ "epoch": 3.389830508474576,
391
+ "grad_norm": 1.8359375,
392
+ "learning_rate": 1.9830858536039742e-06,
393
+ "loss": 7.299415893554688,
394
+ "mean_token_accuracy": 0.025689680464565753,
395
+ "num_tokens": 7025188.0,
396
+ "step": 800
397
+ },
398
+ {
399
+ "entropy": 7.3349973487854,
400
+ "epoch": 3.4957627118644066,
401
+ "grad_norm": 1.90625,
402
+ "learning_rate": 1.524117921870789e-06,
403
+ "loss": 7.336126098632812,
404
+ "mean_token_accuracy": 0.02420500263571739,
405
+ "num_tokens": 7242603.0,
406
+ "step": 825
407
+ },
408
+ {
409
+ "entropy": 7.306003942489624,
410
+ "epoch": 3.601694915254237,
411
+ "grad_norm": 1.765625,
412
+ "learning_rate": 1.121147815976248e-06,
413
+ "loss": 7.2987548828125,
414
+ "mean_token_accuracy": 0.02604618974030018,
415
+ "num_tokens": 7462100.0,
416
+ "step": 850
417
+ },
418
+ {
419
+ "epoch": 3.6779661016949152,
420
+ "eval_entropy": 7.331447899341583,
421
+ "eval_loss": 7.3567795753479,
422
+ "eval_mean_token_accuracy": 0.02568032095829646,
423
+ "eval_num_tokens": 7621408.0,
424
+ "eval_runtime": 7.4184,
425
+ "eval_samples_per_second": 53.65,
426
+ "eval_steps_per_second": 3.37,
427
+ "step": 868
428
+ },
429
+ {
430
+ "entropy": 7.314174528121948,
431
+ "epoch": 3.707627118644068,
432
+ "grad_norm": 2.0,
433
+ "learning_rate": 7.76837848774642e-07,
434
+ "loss": 7.314702758789062,
435
+ "mean_token_accuracy": 0.02541963815689087,
436
+ "num_tokens": 7681092.0,
437
+ "step": 875
438
+ },
439
+ {
440
+ "entropy": 7.335706939697266,
441
+ "epoch": 3.8135593220338984,
442
+ "grad_norm": 2.03125,
443
+ "learning_rate": 4.934627816890469e-07,
444
+ "loss": 7.334556884765625,
445
+ "mean_token_accuracy": 0.02444286733865738,
446
+ "num_tokens": 7902109.0,
447
+ "step": 900
448
+ },
449
+ {
450
+ "entropy": 7.30207275390625,
451
+ "epoch": 3.919491525423729,
452
+ "grad_norm": 1.96875,
453
+ "learning_rate": 2.728947959871353e-07,
454
+ "loss": 7.29387451171875,
455
+ "mean_token_accuracy": 0.02536882445216179,
456
+ "num_tokens": 8122890.0,
457
+ "step": 925
458
+ },
459
+ {
460
+ "entropy": 7.312922801971435,
461
+ "epoch": 4.02542372881356,
462
+ "grad_norm": 1.984375,
463
+ "learning_rate": 1.1659112379354575e-07,
464
+ "loss": 7.30086181640625,
465
+ "mean_token_accuracy": 0.025601605214178563,
466
+ "num_tokens": 8341879.0,
467
+ "step": 950
468
+ },
469
+ {
470
+ "entropy": 7.326199951171875,
471
+ "epoch": 4.13135593220339,
472
+ "grad_norm": 1.953125,
473
+ "learning_rate": 2.558442055732524e-08,
474
+ "loss": 7.32387939453125,
475
+ "mean_token_accuracy": 0.024802666790783405,
476
+ "num_tokens": 8564759.0,
477
+ "step": 975
478
+ },
479
+ {
480
+ "epoch": 4.203389830508475,
481
+ "eval_entropy": 7.331596851348877,
482
+ "eval_loss": 7.356758117675781,
483
+ "eval_mean_token_accuracy": 0.02577025055264433,
484
+ "eval_num_tokens": 8715260.0,
485
+ "eval_runtime": 7.397,
486
+ "eval_samples_per_second": 53.806,
487
+ "eval_steps_per_second": 3.38,
488
+ "step": 992
489
+ }
490
+ ],
491
+ "logging_steps": 25,
492
+ "max_steps": 996,
493
+ "num_input_tokens_seen": 0,
494
+ "num_train_epochs": 5,
495
+ "save_steps": 124,
496
+ "stateful_callbacks": {
497
+ "TrainerControl": {
498
+ "args": {
499
+ "should_epoch_stop": false,
500
+ "should_evaluate": false,
501
+ "should_log": false,
502
+ "should_save": true,
503
+ "should_training_stop": false
504
+ },
505
+ "attributes": {}
506
+ }
507
+ },
508
+ "total_flos": 4.362669184253952e+16,
509
+ "train_batch_size": 16,
510
+ "trial_name": null,
511
+ "trial_params": null
512
+ }
checkpoint-992/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51fb563566aef1daee880df75431bb65187482b5a42194bffca0aab31b9931d2
3
+ size 5368
checkpoint-996/chat_template.jinja ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0]['role'] == 'system' %}
4
+ {{- messages[0]['content'] }}
5
+ {%- else %}
6
+ {{- 'You are Qwen, created by Alibaba Cloud. You are a helpful assistant.' }}
7
+ {%- endif %}
8
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
9
+ {%- for tool in tools %}
10
+ {{- "\n" }}
11
+ {{- tool | tojson }}
12
+ {%- endfor %}
13
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
14
+ {%- else %}
15
+ {%- if messages[0]['role'] == 'system' %}
16
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
17
+ {%- else %}
18
+ {{- '<|im_start|>system\nYou are Qwen, created by Alibaba Cloud. You are a helpful assistant.<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+ {%- for message in messages %}
22
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
23
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
24
+ {%- elif message.role == "assistant" %}
25
+ {{- '<|im_start|>' + message.role }}
26
+ {%- if message.content %}
27
+ {{- '\n' + message.content }}
28
+ {%- endif %}
29
+ {%- for tool_call in message.tool_calls %}
30
+ {%- if tool_call.function is defined %}
31
+ {%- set tool_call = tool_call.function %}
32
+ {%- endif %}
33
+ {{- '\n<tool_call>\n{"name": "' }}
34
+ {{- tool_call.name }}
35
+ {{- '", "arguments": ' }}
36
+ {{- tool_call.arguments | tojson }}
37
+ {{- '}\n</tool_call>' }}
38
+ {%- endfor %}
39
+ {{- '<|im_end|>\n' }}
40
+ {%- elif message.role == "tool" %}
41
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
42
+ {{- '<|im_start|>user' }}
43
+ {%- endif %}
44
+ {{- '\n<tool_response>\n' }}
45
+ {{- message.content }}
46
+ {{- '\n</tool_response>' }}
47
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
48
+ {{- '<|im_end|>\n' }}
49
+ {%- endif %}
50
+ {%- endif %}
51
+ {%- endfor %}
52
+ {%- if add_generation_prompt %}
53
+ {{- '<|im_start|>assistant\n' }}
54
+ {%- endif %}
checkpoint-996/config.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": null,
7
+ "dtype": "bfloat16",
8
+ "eos_token_id": 151645,
9
+ "hidden_act": "silu",
10
+ "hidden_size": 896,
11
+ "initializer_range": 0.02,
12
+ "intermediate_size": 4864,
13
+ "layer_types": [
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention"
38
+ ],
39
+ "max_position_embeddings": 32768,
40
+ "max_window_layers": 21,
41
+ "model_type": "qwen2",
42
+ "num_attention_heads": 14,
43
+ "num_hidden_layers": 24,
44
+ "num_key_value_heads": 2,
45
+ "pad_token_id": 151645,
46
+ "rms_norm_eps": 1e-06,
47
+ "rope_parameters": {
48
+ "rope_theta": 1000000.0,
49
+ "rope_type": "default"
50
+ },
51
+ "sliding_window": null,
52
+ "tie_word_embeddings": true,
53
+ "transformers_version": "5.14.1",
54
+ "use_cache": false,
55
+ "use_sliding_window": false,
56
+ "vocab_size": 217653
57
+ }
checkpoint-996/generation_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_sample": true,
3
+ "eos_token_id": [
4
+ 151645,
5
+ 151643
6
+ ],
7
+ "pad_token_id": 151645,
8
+ "repetition_penalty": 1.1,
9
+ "temperature": 0.7,
10
+ "top_k": 20,
11
+ "top_p": 0.8,
12
+ "transformers_version": "5.14.1"
13
+ }
checkpoint-996/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e6392edcbdb6cf52f7d5325d0c94ba1fe422eabad068967cb4a397130ad311f
3
+ size 1105862784
checkpoint-996/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d7152b28c8dadbf61b30245efe964ffa3457b550ca4087b7c74b0057999e24c
3
+ size 2211903930
checkpoint-996/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8d3b7102895eb0637b0cab516bd672f216b2bf79078a83eb301011a90444f44c
3
+ size 14244
checkpoint-996/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3600ec377e6f8c3453f55e9b7986a3a096c5883da7e8b8e641766a1855e674e2
3
+ size 1064
checkpoint-996/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad19af06e958a8bdd6b45bfbba8a683aa333fc56623bd35f3a3c2b1d9c2a103e
3
+ size 24140418
checkpoint-996/tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "is_local": false,
9
+ "local_files_only": false,
10
+ "model_max_length": 2048,
11
+ "pad_token": "<|im_end|>",
12
+ "padding_side": "right",
13
+ "split_special_tokens": false,
14
+ "tokenizer_class": "Qwen2Tokenizer",
15
+ "unk_token": null
16
+ }
checkpoint-996/trainer_state.json ADDED
@@ -0,0 +1,523 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 992,
3
+ "best_metric": 7.356758117675781,
4
+ "best_model_checkpoint": "/workspace/neutts_air_urdu_finetune/checkpoint-992",
5
+ "epoch": 4.220338983050848,
6
+ "eval_steps": 124,
7
+ "global_step": 996,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 7.697032804489136,
14
+ "epoch": 0.1059322033898305,
15
+ "grad_norm": 4.78125,
16
+ "learning_rate": 1.6000000000000003e-05,
17
+ "loss": 7.989871215820313,
18
+ "mean_token_accuracy": 0.016311284080147745,
19
+ "num_tokens": 218322.0,
20
+ "step": 25
21
+ },
22
+ {
23
+ "entropy": 7.771375341415405,
24
+ "epoch": 0.211864406779661,
25
+ "grad_norm": 2.671875,
26
+ "learning_rate": 1.9980915336317713e-05,
27
+ "loss": 7.803304443359375,
28
+ "mean_token_accuracy": 0.018622982166707516,
29
+ "num_tokens": 437349.0,
30
+ "step": 50
31
+ },
32
+ {
33
+ "entropy": 7.646108131408692,
34
+ "epoch": 0.3177966101694915,
35
+ "grad_norm": 2.421875,
36
+ "learning_rate": 1.989779323028575e-05,
37
+ "loss": 7.629185180664063,
38
+ "mean_token_accuracy": 0.021061680242419244,
39
+ "num_tokens": 658790.0,
40
+ "step": 75
41
+ },
42
+ {
43
+ "entropy": 7.555726566314697,
44
+ "epoch": 0.423728813559322,
45
+ "grad_norm": 2.359375,
46
+ "learning_rate": 1.9749279121818235e-05,
47
+ "loss": 7.55064208984375,
48
+ "mean_token_accuracy": 0.022715183552354575,
49
+ "num_tokens": 872825.0,
50
+ "step": 100
51
+ },
52
+ {
53
+ "epoch": 0.5254237288135594,
54
+ "eval_entropy": 7.486619492371877,
55
+ "eval_loss": 7.498717784881592,
56
+ "eval_mean_token_accuracy": 0.02434113745888074,
57
+ "eval_num_tokens": 1084707.0,
58
+ "eval_runtime": 7.3619,
59
+ "eval_samples_per_second": 54.062,
60
+ "eval_steps_per_second": 3.396,
61
+ "step": 124
62
+ },
63
+ {
64
+ "entropy": 7.509912757873535,
65
+ "epoch": 0.5296610169491526,
66
+ "grad_norm": 2.21875,
67
+ "learning_rate": 1.9536354202855306e-05,
68
+ "loss": 7.5065460205078125,
69
+ "mean_token_accuracy": 0.022764644995331765,
70
+ "num_tokens": 1093864.0,
71
+ "step": 125
72
+ },
73
+ {
74
+ "entropy": 7.504309902191162,
75
+ "epoch": 0.635593220338983,
76
+ "grad_norm": 2.078125,
77
+ "learning_rate": 1.9260425209878052e-05,
78
+ "loss": 7.493028564453125,
79
+ "mean_token_accuracy": 0.022260171882808207,
80
+ "num_tokens": 1315198.0,
81
+ "step": 150
82
+ },
83
+ {
84
+ "entropy": 7.464480457305908,
85
+ "epoch": 0.7415254237288136,
86
+ "grad_norm": 2.15625,
87
+ "learning_rate": 1.8923315129986838e-05,
88
+ "loss": 7.463182373046875,
89
+ "mean_token_accuracy": 0.023391987532377242,
90
+ "num_tokens": 1533213.0,
91
+ "step": 175
92
+ },
93
+ {
94
+ "entropy": 7.452612771987915,
95
+ "epoch": 0.847457627118644,
96
+ "grad_norm": 2.34375,
97
+ "learning_rate": 1.8527251156925997e-05,
98
+ "loss": 7.447894897460937,
99
+ "mean_token_accuracy": 0.02403868570923805,
100
+ "num_tokens": 1752307.0,
101
+ "step": 200
102
+ },
103
+ {
104
+ "entropy": 7.409109563827514,
105
+ "epoch": 0.9533898305084746,
106
+ "grad_norm": 1.96875,
107
+ "learning_rate": 1.8074849976626273e-05,
108
+ "loss": 7.407470703125,
109
+ "mean_token_accuracy": 0.023314249143004416,
110
+ "num_tokens": 1973350.0,
111
+ "step": 225
112
+ },
113
+ {
114
+ "epoch": 1.0508474576271187,
115
+ "eval_entropy": 7.401865323384603,
116
+ "eval_loss": 7.414607524871826,
117
+ "eval_mean_token_accuracy": 0.025272298449029524,
118
+ "eval_num_tokens": 2178943.0,
119
+ "eval_runtime": 7.4146,
120
+ "eval_samples_per_second": 53.678,
121
+ "eval_steps_per_second": 3.372,
122
+ "step": 248
123
+ },
124
+ {
125
+ "entropy": 7.410350885391235,
126
+ "epoch": 1.0593220338983051,
127
+ "grad_norm": 2.078125,
128
+ "learning_rate": 1.756910047947926e-05,
129
+ "loss": 7.405702514648437,
130
+ "mean_token_accuracy": 0.024052796624600886,
131
+ "num_tokens": 2195473.0,
132
+ "step": 250
133
+ },
134
+ {
135
+ "entropy": 7.378520240783692,
136
+ "epoch": 1.1652542372881356,
137
+ "grad_norm": 2.0625,
138
+ "learning_rate": 1.7013344013559197e-05,
139
+ "loss": 7.368658447265625,
140
+ "mean_token_accuracy": 0.024010552018880846,
141
+ "num_tokens": 2410539.0,
142
+ "step": 275
143
+ },
144
+ {
145
+ "entropy": 7.370533514022827,
146
+ "epoch": 1.271186440677966,
147
+ "grad_norm": 2.015625,
148
+ "learning_rate": 1.64112523092536e-05,
149
+ "loss": 7.370477905273438,
150
+ "mean_token_accuracy": 0.02511810462921858,
151
+ "num_tokens": 2630347.0,
152
+ "step": 300
153
+ },
154
+ {
155
+ "entropy": 7.372100067138672,
156
+ "epoch": 1.3771186440677967,
157
+ "grad_norm": 2.140625,
158
+ "learning_rate": 1.5766803221148676e-05,
159
+ "loss": 7.374469604492187,
160
+ "mean_token_accuracy": 0.02450455617159605,
161
+ "num_tokens": 2848534.0,
162
+ "step": 325
163
+ },
164
+ {
165
+ "entropy": 7.342970514297486,
166
+ "epoch": 1.4830508474576272,
167
+ "grad_norm": 1.9609375,
168
+ "learning_rate": 1.5084254447436169e-05,
169
+ "loss": 7.339466552734375,
170
+ "mean_token_accuracy": 0.025138991102576256,
171
+ "num_tokens": 3070123.0,
172
+ "step": 350
173
+ },
174
+ {
175
+ "epoch": 1.576271186440678,
176
+ "eval_entropy": 7.361576656500499,
177
+ "eval_loss": 7.381193161010742,
178
+ "eval_mean_token_accuracy": 0.025545974417279165,
179
+ "eval_num_tokens": 3263577.0,
180
+ "eval_runtime": 7.4123,
181
+ "eval_samples_per_second": 53.695,
182
+ "eval_steps_per_second": 3.373,
183
+ "step": 372
184
+ },
185
+ {
186
+ "entropy": 7.361075611114502,
187
+ "epoch": 1.5889830508474576,
188
+ "grad_norm": 1.9140625,
189
+ "learning_rate": 1.4368115400470393e-05,
190
+ "loss": 7.3520880126953125,
191
+ "mean_token_accuracy": 0.024180141538381578,
192
+ "num_tokens": 3288806.0,
193
+ "step": 375
194
+ },
195
+ {
196
+ "entropy": 7.3546325302124025,
197
+ "epoch": 1.694915254237288,
198
+ "grad_norm": 2.15625,
199
+ "learning_rate": 1.3623117414318827e-05,
200
+ "loss": 7.346293334960937,
201
+ "mean_token_accuracy": 0.024999124109745027,
202
+ "num_tokens": 3510776.0,
203
+ "step": 400
204
+ },
205
+ {
206
+ "entropy": 7.363360710144043,
207
+ "epoch": 1.8008474576271185,
208
+ "grad_norm": 2.078125,
209
+ "learning_rate": 1.2854182486136944e-05,
210
+ "loss": 7.364286499023438,
211
+ "mean_token_accuracy": 0.02418241061270237,
212
+ "num_tokens": 3731405.0,
213
+ "step": 425
214
+ },
215
+ {
216
+ "entropy": 7.334566602706909,
217
+ "epoch": 1.9067796610169492,
218
+ "grad_norm": 1.9609375,
219
+ "learning_rate": 1.2066390757884328e-05,
220
+ "loss": 7.33093505859375,
221
+ "mean_token_accuracy": 0.02495731335133314,
222
+ "num_tokens": 3953207.0,
223
+ "step": 450
224
+ },
225
+ {
226
+ "entropy": 7.353079776763916,
227
+ "epoch": 2.01271186440678,
228
+ "grad_norm": 1.8671875,
229
+ "learning_rate": 1.1264946953221496e-05,
230
+ "loss": 7.351827392578125,
231
+ "mean_token_accuracy": 0.02447190221399069,
232
+ "num_tokens": 4169521.0,
233
+ "step": 475
234
+ },
235
+ {
236
+ "epoch": 2.1016949152542375,
237
+ "eval_entropy": 7.34862866004308,
238
+ "eval_loss": 7.365415096282959,
239
+ "eval_mean_token_accuracy": 0.025626841001212597,
240
+ "eval_num_tokens": 4351452.0,
241
+ "eval_runtime": 7.4142,
242
+ "eval_samples_per_second": 53.681,
243
+ "eval_steps_per_second": 3.372,
244
+ "step": 496
245
+ },
246
+ {
247
+ "entropy": 7.339797382354736,
248
+ "epoch": 2.1186440677966103,
249
+ "grad_norm": 2.03125,
250
+ "learning_rate": 1.0455145991329639e-05,
251
+ "loss": 7.3259820556640625,
252
+ "mean_token_accuracy": 0.025096801929175853,
253
+ "num_tokens": 4386912.0,
254
+ "step": 500
255
+ },
256
+ {
257
+ "entropy": 7.330105533599854,
258
+ "epoch": 2.2245762711864407,
259
+ "grad_norm": 2.03125,
260
+ "learning_rate": 9.642338004833295e-06,
261
+ "loss": 7.330958251953125,
262
+ "mean_token_accuracy": 0.024473249688744547,
263
+ "num_tokens": 4607819.0,
264
+ "step": 525
265
+ },
266
+ {
267
+ "entropy": 7.331509189605713,
268
+ "epoch": 2.330508474576271,
269
+ "grad_norm": 1.96875,
270
+ "learning_rate": 8.831892992943e-06,
271
+ "loss": 7.329585571289062,
272
+ "mean_token_accuracy": 0.024803164564073086,
273
+ "num_tokens": 4829926.0,
274
+ "step": 550
275
+ },
276
+ {
277
+ "entropy": 7.332346143722535,
278
+ "epoch": 2.4364406779661016,
279
+ "grad_norm": 1.953125,
280
+ "learning_rate": 8.029165343344805e-06,
281
+ "loss": 7.3258428955078125,
282
+ "mean_token_accuracy": 0.02469826426357031,
283
+ "num_tokens": 5045115.0,
284
+ "step": 575
285
+ },
286
+ {
287
+ "entropy": 7.318505411148071,
288
+ "epoch": 2.542372881355932,
289
+ "grad_norm": 1.9140625,
290
+ "learning_rate": 7.2394584572309125e-06,
291
+ "loss": 7.3118963623046875,
292
+ "mean_token_accuracy": 0.02518722042441368,
293
+ "num_tokens": 5265448.0,
294
+ "step": 600
295
+ },
296
+ {
297
+ "epoch": 2.6271186440677967,
298
+ "eval_entropy": 7.333844860394795,
299
+ "eval_loss": 7.359093189239502,
300
+ "eval_mean_token_accuracy": 0.025550531456246972,
301
+ "eval_num_tokens": 5439386.0,
302
+ "eval_runtime": 7.4052,
303
+ "eval_samples_per_second": 53.746,
304
+ "eval_steps_per_second": 3.376,
305
+ "step": 620
306
+ },
307
+ {
308
+ "entropy": 7.321087207794189,
309
+ "epoch": 2.648305084745763,
310
+ "grad_norm": 2.0625,
311
+ "learning_rate": 6.467989711184021e-06,
312
+ "loss": 7.31301513671875,
313
+ "mean_token_accuracy": 0.025983325839042663,
314
+ "num_tokens": 5482207.0,
315
+ "step": 625
316
+ },
317
+ {
318
+ "entropy": 7.322469959259033,
319
+ "epoch": 2.7542372881355934,
320
+ "grad_norm": 2.125,
321
+ "learning_rate": 5.7198559874026945e-06,
322
+ "loss": 7.318033447265625,
323
+ "mean_token_accuracy": 0.02565582599490881,
324
+ "num_tokens": 5703727.0,
325
+ "step": 650
326
+ },
327
+ {
328
+ "entropy": 7.31039176940918,
329
+ "epoch": 2.860169491525424,
330
+ "grad_norm": 2.03125,
331
+ "learning_rate": 5.000000000000003e-06,
332
+ "loss": 7.305034790039063,
333
+ "mean_token_accuracy": 0.02535436548292637,
334
+ "num_tokens": 5923157.0,
335
+ "step": 675
336
+ },
337
+ {
338
+ "entropy": 7.3113681411743165,
339
+ "epoch": 2.9661016949152543,
340
+ "grad_norm": 1.9765625,
341
+ "learning_rate": 4.313177639848408e-06,
342
+ "loss": 7.3114166259765625,
343
+ "mean_token_accuracy": 0.025559407025575638,
344
+ "num_tokens": 6143779.0,
345
+ "step": 700
346
+ },
347
+ {
348
+ "entropy": 7.3387535953521725,
349
+ "epoch": 3.0720338983050848,
350
+ "grad_norm": 1.921875,
351
+ "learning_rate": 3.6639265537145187e-06,
352
+ "loss": 7.343036499023437,
353
+ "mean_token_accuracy": 0.024094666354358196,
354
+ "num_tokens": 6365371.0,
355
+ "step": 725
356
+ },
357
+ {
358
+ "epoch": 3.152542372881356,
359
+ "eval_entropy": 7.332533776760101,
360
+ "eval_loss": 7.357120990753174,
361
+ "eval_mean_token_accuracy": 0.025878646954273183,
362
+ "eval_num_tokens": 6532337.0,
363
+ "eval_runtime": 7.4033,
364
+ "eval_samples_per_second": 53.76,
365
+ "eval_steps_per_second": 3.377,
366
+ "step": 744
367
+ },
368
+ {
369
+ "entropy": 7.320968351364136,
370
+ "epoch": 3.1779661016949152,
371
+ "grad_norm": 2.109375,
372
+ "learning_rate": 3.056536165272662e-06,
373
+ "loss": 7.313834228515625,
374
+ "mean_token_accuracy": 0.025647504702210427,
375
+ "num_tokens": 6584143.0,
376
+ "step": 750
377
+ },
378
+ {
379
+ "entropy": 7.319151067733765,
380
+ "epoch": 3.2838983050847457,
381
+ "grad_norm": 1.9140625,
382
+ "learning_rate": 2.4950193360603868e-06,
383
+ "loss": 7.315574951171875,
384
+ "mean_token_accuracy": 0.02516275253146887,
385
+ "num_tokens": 6806162.0,
386
+ "step": 775
387
+ },
388
+ {
389
+ "entropy": 7.312239484786987,
390
+ "epoch": 3.389830508474576,
391
+ "grad_norm": 1.8359375,
392
+ "learning_rate": 1.9830858536039742e-06,
393
+ "loss": 7.299415893554688,
394
+ "mean_token_accuracy": 0.025689680464565753,
395
+ "num_tokens": 7025188.0,
396
+ "step": 800
397
+ },
398
+ {
399
+ "entropy": 7.3349973487854,
400
+ "epoch": 3.4957627118644066,
401
+ "grad_norm": 1.90625,
402
+ "learning_rate": 1.524117921870789e-06,
403
+ "loss": 7.336126098632812,
404
+ "mean_token_accuracy": 0.02420500263571739,
405
+ "num_tokens": 7242603.0,
406
+ "step": 825
407
+ },
408
+ {
409
+ "entropy": 7.306003942489624,
410
+ "epoch": 3.601694915254237,
411
+ "grad_norm": 1.765625,
412
+ "learning_rate": 1.121147815976248e-06,
413
+ "loss": 7.2987548828125,
414
+ "mean_token_accuracy": 0.02604618974030018,
415
+ "num_tokens": 7462100.0,
416
+ "step": 850
417
+ },
418
+ {
419
+ "epoch": 3.6779661016949152,
420
+ "eval_entropy": 7.331447899341583,
421
+ "eval_loss": 7.3567795753479,
422
+ "eval_mean_token_accuracy": 0.02568032095829646,
423
+ "eval_num_tokens": 7621408.0,
424
+ "eval_runtime": 7.4184,
425
+ "eval_samples_per_second": 53.65,
426
+ "eval_steps_per_second": 3.37,
427
+ "step": 868
428
+ },
429
+ {
430
+ "entropy": 7.314174528121948,
431
+ "epoch": 3.707627118644068,
432
+ "grad_norm": 2.0,
433
+ "learning_rate": 7.76837848774642e-07,
434
+ "loss": 7.314702758789062,
435
+ "mean_token_accuracy": 0.02541963815689087,
436
+ "num_tokens": 7681092.0,
437
+ "step": 875
438
+ },
439
+ {
440
+ "entropy": 7.335706939697266,
441
+ "epoch": 3.8135593220338984,
442
+ "grad_norm": 2.03125,
443
+ "learning_rate": 4.934627816890469e-07,
444
+ "loss": 7.334556884765625,
445
+ "mean_token_accuracy": 0.02444286733865738,
446
+ "num_tokens": 7902109.0,
447
+ "step": 900
448
+ },
449
+ {
450
+ "entropy": 7.30207275390625,
451
+ "epoch": 3.919491525423729,
452
+ "grad_norm": 1.96875,
453
+ "learning_rate": 2.728947959871353e-07,
454
+ "loss": 7.29387451171875,
455
+ "mean_token_accuracy": 0.02536882445216179,
456
+ "num_tokens": 8122890.0,
457
+ "step": 925
458
+ },
459
+ {
460
+ "entropy": 7.312922801971435,
461
+ "epoch": 4.02542372881356,
462
+ "grad_norm": 1.984375,
463
+ "learning_rate": 1.1659112379354575e-07,
464
+ "loss": 7.30086181640625,
465
+ "mean_token_accuracy": 0.025601605214178563,
466
+ "num_tokens": 8341879.0,
467
+ "step": 950
468
+ },
469
+ {
470
+ "entropy": 7.326199951171875,
471
+ "epoch": 4.13135593220339,
472
+ "grad_norm": 1.953125,
473
+ "learning_rate": 2.558442055732524e-08,
474
+ "loss": 7.32387939453125,
475
+ "mean_token_accuracy": 0.024802666790783405,
476
+ "num_tokens": 8564759.0,
477
+ "step": 975
478
+ },
479
+ {
480
+ "epoch": 4.203389830508475,
481
+ "eval_entropy": 7.331596851348877,
482
+ "eval_loss": 7.356758117675781,
483
+ "eval_mean_token_accuracy": 0.02577025055264433,
484
+ "eval_num_tokens": 8715260.0,
485
+ "eval_runtime": 7.397,
486
+ "eval_samples_per_second": 53.806,
487
+ "eval_steps_per_second": 3.38,
488
+ "step": 992
489
+ },
490
+ {
491
+ "epoch": 4.220338983050848,
492
+ "eval_entropy": 7.331597983837128,
493
+ "eval_loss": 7.3567633628845215,
494
+ "eval_mean_token_accuracy": 0.02577025055264433,
495
+ "eval_num_tokens": 8750612.0,
496
+ "eval_runtime": 7.4178,
497
+ "eval_samples_per_second": 53.655,
498
+ "eval_steps_per_second": 3.37,
499
+ "step": 996
500
+ }
501
+ ],
502
+ "logging_steps": 25,
503
+ "max_steps": 996,
504
+ "num_input_tokens_seen": 0,
505
+ "num_train_epochs": 5,
506
+ "save_steps": 124,
507
+ "stateful_callbacks": {
508
+ "TrainerControl": {
509
+ "args": {
510
+ "should_epoch_stop": false,
511
+ "should_evaluate": false,
512
+ "should_log": false,
513
+ "should_save": true,
514
+ "should_training_stop": true
515
+ },
516
+ "attributes": {}
517
+ }
518
+ },
519
+ "total_flos": 4.380260592254976e+16,
520
+ "train_batch_size": 16,
521
+ "trial_name": null,
522
+ "trial_params": null
523
+ }
checkpoint-996/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51fb563566aef1daee880df75431bb65187482b5a42194bffca0aab31b9931d2
3
+ size 5368
config.json ADDED
@@ -0,0 +1,57 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2ForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": null,
7
+ "dtype": "bfloat16",
8
+ "eos_token_id": 151645,
9
+ "hidden_act": "silu",
10
+ "hidden_size": 896,
11
+ "initializer_range": 0.02,
12
+ "intermediate_size": 4864,
13
+ "layer_types": [
14
+ "full_attention",
15
+ "full_attention",
16
+ "full_attention",
17
+ "full_attention",
18
+ "full_attention",
19
+ "full_attention",
20
+ "full_attention",
21
+ "full_attention",
22
+ "full_attention",
23
+ "full_attention",
24
+ "full_attention",
25
+ "full_attention",
26
+ "full_attention",
27
+ "full_attention",
28
+ "full_attention",
29
+ "full_attention",
30
+ "full_attention",
31
+ "full_attention",
32
+ "full_attention",
33
+ "full_attention",
34
+ "full_attention",
35
+ "full_attention",
36
+ "full_attention",
37
+ "full_attention"
38
+ ],
39
+ "max_position_embeddings": 32768,
40
+ "max_window_layers": 21,
41
+ "model_type": "qwen2",
42
+ "num_attention_heads": 14,
43
+ "num_hidden_layers": 24,
44
+ "num_key_value_heads": 2,
45
+ "pad_token_id": 151645,
46
+ "rms_norm_eps": 1e-06,
47
+ "rope_parameters": {
48
+ "rope_theta": 1000000.0,
49
+ "rope_type": "default"
50
+ },
51
+ "sliding_window": null,
52
+ "tie_word_embeddings": true,
53
+ "transformers_version": "5.14.1",
54
+ "use_cache": false,
55
+ "use_sliding_window": false,
56
+ "vocab_size": 217653
57
+ }
generation_config.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_sample": true,
3
+ "eos_token_id": [
4
+ 151645,
5
+ 151643
6
+ ],
7
+ "pad_token_id": 151645,
8
+ "repetition_penalty": 1.1,
9
+ "temperature": 0.7,
10
+ "top_k": 20,
11
+ "top_p": 0.8,
12
+ "transformers_version": "5.14.1"
13
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6217e5cd2f1fcb6b5775d40c6af19bef77ccfc0785719c2a3fea3c27f9d046f8
3
+ size 1105862784
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ad19af06e958a8bdd6b45bfbba8a683aa333fc56623bd35f3a3c2b1d9c2a103e
3
+ size 24140418
tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "is_local": false,
9
+ "local_files_only": false,
10
+ "model_max_length": 2048,
11
+ "pad_token": "<|im_end|>",
12
+ "padding_side": "right",
13
+ "split_special_tokens": false,
14
+ "tokenizer_class": "Qwen2Tokenizer",
15
+ "unk_token": null
16
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51fb563566aef1daee880df75431bb65187482b5a42194bffca0aab31b9931d2
3
+ size 5368