diff --git a/.gitattributes b/.gitattributes index a6344aac8c09253b3b630fb776ae94478aa0275b..757bf8cf8f1f41236b8c5e9f1691db07111bf76e 100644 --- a/.gitattributes +++ b/.gitattributes @@ -33,3 +33,42 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text *.zip filter=lfs diff=lfs merge=lfs -text *.zst filter=lfs diff=lfs merge=lfs -text *tfevents* filter=lfs diff=lfs merge=lfs -text +checkpoint-1000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-1200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-1400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-1600/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-1800/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-2000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-2200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-2400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-2600/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-2800/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-3000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-3200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-3400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-3600/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-3800/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-4000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-4200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-4400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-4600/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-4800/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-5000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-5200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-5400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-5600/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-5800/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-600/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-6000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-6200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-6400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-6600/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-6800/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-7000/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-7200/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-7400/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-7460/tokenizer.json filter=lfs diff=lfs merge=lfs -text +checkpoint-800/tokenizer.json filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/adapter_config.json b/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/adapter_model.safetensors b/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2d56dd52e04576c533bceb18ec394b50943e9d91 --- /dev/null +++ b/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e45b98549709b9485f32de70adf0f96edce8ca97b599382c87a97973290a9c87 +size 50360752 diff --git a/chat_template.jinja b/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-1000/README.md b/checkpoint-1000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-1000/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-1000/adapter_config.json b/checkpoint-1000/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-1000/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-1000/adapter_model.safetensors b/checkpoint-1000/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e74367298f28c1e41eb82bf2fcb3e7da4b0fc005 --- /dev/null +++ b/checkpoint-1000/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4ac3894d523e35894aeae2b1c086e01fcae4ba05069b56b46969eb0f2504c8e6 +size 50360752 diff --git a/checkpoint-1000/chat_template.jinja b/checkpoint-1000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-1000/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-1000/optimizer.pt b/checkpoint-1000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..e487d54aaec30206b9956a2e1e8f282392bf3ab4 --- /dev/null +++ b/checkpoint-1000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de1b730d984dc003995b54aff0084253a58ef6b058175c32d84a3d1d1e3c717b +size 100828235 diff --git a/checkpoint-1000/rng_state.pth b/checkpoint-1000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..70b420ea263b5924d82b3936b2173962c460358f --- /dev/null +++ b/checkpoint-1000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b8c78882d5b2c84736c1873ea3d414ad53537972adacc7dccf326a9ebc5a0f89 +size 14645 diff --git a/checkpoint-1000/scheduler.pt b/checkpoint-1000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bd33870f5c03449172f9c197eca91cf9b1caa083 --- /dev/null +++ b/checkpoint-1000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f226462606e1b78f7191acb3109491600d12573ce587511117ddff07f83f50dd +size 1465 diff --git a/checkpoint-1000/tokenizer.json b/checkpoint-1000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-1000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-1000/tokenizer_config.json b/checkpoint-1000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-1000/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-1000/trainer_state.json b/checkpoint-1000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e0823e1b9b5c364710ae8f239051dd22193b9722 --- /dev/null +++ b/checkpoint-1000/trainer_state.json @@ -0,0 +1,384 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.26812347085833027, + "eval_steps": 500, + "global_step": 1000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.2265883151887155e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1000/training_args.bin b/checkpoint-1000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-1000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-1200/README.md b/checkpoint-1200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-1200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-1200/adapter_config.json b/checkpoint-1200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-1200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-1200/adapter_model.safetensors b/checkpoint-1200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..c860b0c11b9297970e3b399c4331e9abbff2551f --- /dev/null +++ b/checkpoint-1200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:417989dbb49cc7b8b39ad62db038f967e940507b46ee8c39482d35892346c728 +size 50360752 diff --git a/checkpoint-1200/chat_template.jinja b/checkpoint-1200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-1200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-1200/optimizer.pt b/checkpoint-1200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..55d92805c5ea321f97a89bb4e145e268b5c91858 --- /dev/null +++ b/checkpoint-1200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e27e2c9c771096aa76b05aae3b399be1b383b6a9ba510a9615b0f6f8a6f70119 +size 100828235 diff --git a/checkpoint-1200/rng_state.pth b/checkpoint-1200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..872452357941975936cf2690d425d3e2aa6276ab --- /dev/null +++ b/checkpoint-1200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:96cde9f5973b91631d0681c40dd52cfa938ee26803320fa23d35863f4f5f4217 +size 14645 diff --git a/checkpoint-1200/scheduler.pt b/checkpoint-1200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ecf4058648be9b281ad20aa1f50ad6aa383576d8 --- /dev/null +++ b/checkpoint-1200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5049e5d5d140d2f9add531f3077ff375aff61236ef16509409553782896408a3 +size 1465 diff --git a/checkpoint-1200/tokenizer.json b/checkpoint-1200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-1200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-1200/tokenizer_config.json b/checkpoint-1200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-1200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-1200/trainer_state.json b/checkpoint-1200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d0041a6ecf2b3675b575a7cdb99dbb3e9a2f11b6 --- /dev/null +++ b/checkpoint-1200/trainer_state.json @@ -0,0 +1,454 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.3217481650299963, + "eval_steps": 500, + "global_step": 1200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.471552740097106e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1200/training_args.bin b/checkpoint-1200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-1200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-1400/README.md b/checkpoint-1400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-1400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-1400/adapter_config.json b/checkpoint-1400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-1400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-1400/adapter_model.safetensors b/checkpoint-1400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6ced79fbab838e8e5473dd33bdb1d8b10f9f8b20 --- /dev/null +++ b/checkpoint-1400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4fe4375c944d5df24e8c793700e13aed770284562c56c3a8cc6808b38bd7abc0 +size 50360752 diff --git a/checkpoint-1400/chat_template.jinja b/checkpoint-1400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-1400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-1400/optimizer.pt b/checkpoint-1400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..4dd9f29db24eb422de368e0698fe783f93f685f8 --- /dev/null +++ b/checkpoint-1400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:47d94db2fd46b9422d93e25e91453d20756375ac76aac6663b03b6a6fc25bef4 +size 100828235 diff --git a/checkpoint-1400/rng_state.pth b/checkpoint-1400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4fa50e77b3fe1334d29b3eb91d8a99958af49bbe --- /dev/null +++ b/checkpoint-1400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:162580b2a1cba676dfe8710060a053ba2797d5b8bc3c88d23d7586fbc01083eb +size 14645 diff --git a/checkpoint-1400/scheduler.pt b/checkpoint-1400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..37b468a8cb502aeadfa2cccf9d4f4784e743fb6e --- /dev/null +++ b/checkpoint-1400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8ae33e797bdb8b83fcfd3d3c8acd9fdd0e0c30ff0058439330b84aaf64f957b7 +size 1465 diff --git a/checkpoint-1400/tokenizer.json b/checkpoint-1400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-1400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-1400/tokenizer_config.json b/checkpoint-1400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-1400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-1400/trainer_state.json b/checkpoint-1400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c0eed54cd0f497731de2e24e0743d9ab1673f23b --- /dev/null +++ b/checkpoint-1400/trainer_state.json @@ -0,0 +1,524 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.37537285920166236, + "eval_steps": 500, + "global_step": 1400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.716587745411932e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1400/training_args.bin b/checkpoint-1400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-1400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-1600/README.md b/checkpoint-1600/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-1600/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-1600/adapter_config.json b/checkpoint-1600/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-1600/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-1600/adapter_model.safetensors b/checkpoint-1600/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..99e87d799cc5eff48620b823bf22a276f435f79d --- /dev/null +++ b/checkpoint-1600/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c3c7d20f9c92c83c3e12ff31d71f58e8745aaf3d1f044cb5705d000d14c20c1a +size 50360752 diff --git a/checkpoint-1600/chat_template.jinja b/checkpoint-1600/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-1600/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-1600/optimizer.pt b/checkpoint-1600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7e6dc0a9433d391d36a05423bd2cce992fd3d7f9 --- /dev/null +++ b/checkpoint-1600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad434d84ce693267a0de2e31b4b3d4720932da875afad04190d6e7a5cdf89efc +size 100828235 diff --git a/checkpoint-1600/rng_state.pth b/checkpoint-1600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..0f730a278bfffb68aa4937ec72d8254d55df68b3 --- /dev/null +++ b/checkpoint-1600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:219bdda0c11190b268d7e89b412db53f775797054aa6786905e1d43e958bd9ff +size 14645 diff --git a/checkpoint-1600/scheduler.pt b/checkpoint-1600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..833d3603ffb66bab9f7e86122acad5c41ac4e14e --- /dev/null +++ b/checkpoint-1600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3b564ed9b760aa1e9458ee373fe09c366ab6ead87c7c8a1b22b2df6345a99f67 +size 1465 diff --git a/checkpoint-1600/tokenizer.json b/checkpoint-1600/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-1600/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-1600/tokenizer_config.json b/checkpoint-1600/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-1600/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-1600/trainer_state.json b/checkpoint-1600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c095f74584b4510a4dd075afd7f73771a033f26c --- /dev/null +++ b/checkpoint-1600/trainer_state.json @@ -0,0 +1,594 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.4289975533733284, + "eval_steps": 500, + "global_step": 1600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.9650753089415782e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1600/training_args.bin b/checkpoint-1600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-1600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-1800/README.md b/checkpoint-1800/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-1800/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-1800/adapter_config.json b/checkpoint-1800/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-1800/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-1800/adapter_model.safetensors b/checkpoint-1800/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3a2134f3c62b3ee453c505dae936807ad2e190ff --- /dev/null +++ b/checkpoint-1800/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad650210ad416713c19e4dfcf053b5187aad88eb7af7712cd60a8c9b074573d3 +size 50360752 diff --git a/checkpoint-1800/chat_template.jinja b/checkpoint-1800/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-1800/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-1800/optimizer.pt b/checkpoint-1800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f4cd6a0c8b36f8b5dc0f89189b45db01a2491349 --- /dev/null +++ b/checkpoint-1800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fa831b9206f20a510557e89c9e2492c185d5825c708533098fd94b41ec3ff9a1 +size 100828235 diff --git a/checkpoint-1800/rng_state.pth b/checkpoint-1800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5b08f27b93cc04fc64daab11464d45eeeb2ec5a7 --- /dev/null +++ b/checkpoint-1800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3c5db57a6f231ea08d151da49a2a16729dde52229c213948f11f47d94db8d1ba +size 14645 diff --git a/checkpoint-1800/scheduler.pt b/checkpoint-1800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..acb28cfdf1d66e698d4e5244644182b1964687d6 --- /dev/null +++ b/checkpoint-1800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:abd2865514c1384e677df4f6924dbe82113eabee331604867a3a83cd07c3a494 +size 1465 diff --git a/checkpoint-1800/tokenizer.json b/checkpoint-1800/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-1800/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-1800/tokenizer_config.json b/checkpoint-1800/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-1800/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-1800/trainer_state.json b/checkpoint-1800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..20009ebbdf33c1c019a321d43b215dc1af773396 --- /dev/null +++ b/checkpoint-1800/trainer_state.json @@ -0,0 +1,664 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.48262224754499444, + "eval_steps": 500, + "global_step": 1800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.2117387050620314e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-1800/training_args.bin b/checkpoint-1800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-1800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-200/README.md b/checkpoint-200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-200/adapter_config.json b/checkpoint-200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-200/adapter_model.safetensors b/checkpoint-200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3fdd8483ea567e069f94fc6dd03aa3e3191aef2a --- /dev/null +++ b/checkpoint-200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d237f34f08ec41369ad8edd17a168795fc3671b94230c59bb06dbbe4d98b9eb +size 50360752 diff --git a/checkpoint-200/chat_template.jinja b/checkpoint-200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-200/optimizer.pt b/checkpoint-200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..255da496fea8cc23c80e3bcac42407aaadb45519 --- /dev/null +++ b/checkpoint-200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:538c31e3f2f0229d802a70a0379ac1ca3aaa6fe67be0d5827c4b8fd6d1f17ed1 +size 100828235 diff --git a/checkpoint-200/rng_state.pth b/checkpoint-200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..dab0e2f4f62aa0f90488b90a5bfc4f4854401576 --- /dev/null +++ b/checkpoint-200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:45dc76167e6efb068117728fc0ca9c4f1c97428a1b876b6cf7fd1d0bb4d198f2 +size 14645 diff --git a/checkpoint-200/scheduler.pt b/checkpoint-200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..6afb91e10fce82d6a4fc32a1a5ed0a2530936cef --- /dev/null +++ b/checkpoint-200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:01a6f3499d0acb3d598cd8b7037c45927c9a1d5f189a53dc2d71d0774aaba8ef +size 1465 diff --git a/checkpoint-200/tokenizer.json b/checkpoint-200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-200/tokenizer_config.json b/checkpoint-200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-200/trainer_state.json b/checkpoint-200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6551aeb99d64cc8e4bbc5f3ad079ad4b78218588 --- /dev/null +++ b/checkpoint-200/trainer_state.json @@ -0,0 +1,104 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.05362469417166605, + "eval_steps": 500, + "global_step": 200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.461979015351501e+16, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-200/training_args.bin b/checkpoint-200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-2000/README.md b/checkpoint-2000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-2000/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-2000/adapter_config.json b/checkpoint-2000/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-2000/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-2000/adapter_model.safetensors b/checkpoint-2000/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..00855b76cfc24c5c07c71308e10b64990ad85a36 --- /dev/null +++ b/checkpoint-2000/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:91c1d9072dd3bc5619ee8538cea97d04eea3180a80350e6f16fc8e1de6d46681 +size 50360752 diff --git a/checkpoint-2000/chat_template.jinja b/checkpoint-2000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-2000/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-2000/optimizer.pt b/checkpoint-2000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d5febde6da50172eaef9714d1893eb233f883970 --- /dev/null +++ b/checkpoint-2000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:655f460495b3f9994817435da5aab200c465986092596a6cdbaa165ead1eaf82 +size 100828235 diff --git a/checkpoint-2000/rng_state.pth b/checkpoint-2000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..15c397e4ae186ba8f755a9e3eee8478ffa336091 --- /dev/null +++ b/checkpoint-2000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10bc2585985998753712abc2ecadfff34b6e67963041968dba2cbe27e7cf0c3b +size 14645 diff --git a/checkpoint-2000/scheduler.pt b/checkpoint-2000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..10544ba513751c21a3b7a4f677046a56f6928804 --- /dev/null +++ b/checkpoint-2000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:26794880ede29b4548ba836f84d0aff6b385a2d83360206591e0210dd078bfba +size 1465 diff --git a/checkpoint-2000/tokenizer.json b/checkpoint-2000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-2000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-2000/tokenizer_config.json b/checkpoint-2000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-2000/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-2000/trainer_state.json b/checkpoint-2000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..810731163f73c20b5478109528ac0b0ebcaf0f96 --- /dev/null +++ b/checkpoint-2000/trainer_state.json @@ -0,0 +1,734 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.5362469417166605, + "eval_steps": 500, + "global_step": 2000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.458008027246551e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2000/training_args.bin b/checkpoint-2000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-2000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-2200/README.md b/checkpoint-2200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-2200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-2200/adapter_config.json b/checkpoint-2200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-2200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-2200/adapter_model.safetensors b/checkpoint-2200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ddcf27fa4391a77e069c9bc06a898371057584e5 --- /dev/null +++ b/checkpoint-2200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c09137fa3c09d45e0ddaf3d37db926e20af9dc0d2d013715031618fa885a0148 +size 50360752 diff --git a/checkpoint-2200/chat_template.jinja b/checkpoint-2200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-2200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-2200/optimizer.pt b/checkpoint-2200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..091e3e241e2ac36cbd21647ba4849703f1f4e098 --- /dev/null +++ b/checkpoint-2200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e34a7eafa5ffe9d595da548631104b7569a3c69d14d9203e2b0b697d0a2c028 +size 100828235 diff --git a/checkpoint-2200/rng_state.pth b/checkpoint-2200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..203f470eceea2585413ceb87af9bf57b36dfb364 --- /dev/null +++ b/checkpoint-2200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5ae3ab067567845cbba15e0db0b98e81590592ebcaf63ef8185684a4e0f8715e +size 14645 diff --git a/checkpoint-2200/scheduler.pt b/checkpoint-2200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..e0c42fc5ac5eba1753ddf1ee9351c666ffcda21c --- /dev/null +++ b/checkpoint-2200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3093f2c7ea7a67f370a885013a373c42b22b12bd84fae95a0221e821ecf823e9 +size 1465 diff --git a/checkpoint-2200/tokenizer.json b/checkpoint-2200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-2200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-2200/tokenizer_config.json b/checkpoint-2200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-2200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-2200/trainer_state.json b/checkpoint-2200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f2a91db473979bf0e8aa4adc01ccb99999aa30d0 --- /dev/null +++ b/checkpoint-2200/trainer_state.json @@ -0,0 +1,804 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.5898716358883266, + "eval_steps": 500, + "global_step": 2200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.7065266797647462e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2200/training_args.bin b/checkpoint-2200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-2200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-2400/README.md b/checkpoint-2400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-2400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-2400/adapter_config.json b/checkpoint-2400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-2400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-2400/adapter_model.safetensors b/checkpoint-2400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..63055c8df26b5ba57706bb0c745ca72c62244ff5 --- /dev/null +++ b/checkpoint-2400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:352715f0b39309e4ee6508d2f855d41161303138355eb7548bcc6f459c746896 +size 50360752 diff --git a/checkpoint-2400/chat_template.jinja b/checkpoint-2400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-2400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-2400/optimizer.pt b/checkpoint-2400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7e0b798258393824e04b3b346f495b235fc1eb7a --- /dev/null +++ b/checkpoint-2400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e620d8259ac31856aa62169803645e6d592d7aee6db3b791a105ea999292d6d +size 100828235 diff --git a/checkpoint-2400/rng_state.pth b/checkpoint-2400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..c7857977dbc9a48f9ab58fb852430c8256622ec8 --- /dev/null +++ b/checkpoint-2400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:000f6ba765c74de08e259ce69b3eac86d97620f916260059014e36f6846c8df6 +size 14645 diff --git a/checkpoint-2400/scheduler.pt b/checkpoint-2400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..82ca3ab98e9662810a5e482f3de93f87b42ca4fa --- /dev/null +++ b/checkpoint-2400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1cc087535592c8cec31c3995ac9a31c01df67976c2c9eeb48eaf04a3830becc3 +size 1465 diff --git a/checkpoint-2400/tokenizer.json b/checkpoint-2400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-2400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-2400/tokenizer_config.json b/checkpoint-2400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-2400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-2400/trainer_state.json b/checkpoint-2400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c66f090ee621b717e737170dc06ce8549995bc70 --- /dev/null +++ b/checkpoint-2400/trainer_state.json @@ -0,0 +1,874 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.6434963300599926, + "eval_steps": 500, + "global_step": 2400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.9507768981794406e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2400/training_args.bin b/checkpoint-2400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-2400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-2600/README.md b/checkpoint-2600/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-2600/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-2600/adapter_config.json b/checkpoint-2600/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-2600/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-2600/adapter_model.safetensors b/checkpoint-2600/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7afa743a254be266aca9413216656bfa7383f787 --- /dev/null +++ b/checkpoint-2600/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d06bf39f110778bcb964a43bc1ef290d38366d43018a89d09d3a37e7e696cf57 +size 50360752 diff --git a/checkpoint-2600/chat_template.jinja b/checkpoint-2600/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-2600/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-2600/optimizer.pt b/checkpoint-2600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7514cdb8c6a1ee0d117963db6e89e63e6aeb00c6 --- /dev/null +++ b/checkpoint-2600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c4cd29ec1fcdb0431247aac514fab75ae6888193dc24cd42bb6920b78d722c8d +size 100828235 diff --git a/checkpoint-2600/rng_state.pth b/checkpoint-2600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..32083d5f553aa9ce09923cb77dc9e880bc79adfb --- /dev/null +++ b/checkpoint-2600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:89b752ff6305664845d0a8f87ae70ed79c3c563144db28900d8a172b9dfc9780 +size 14645 diff --git a/checkpoint-2600/scheduler.pt b/checkpoint-2600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..3b492052cdb19375a1d17b7622090617edfdedae --- /dev/null +++ b/checkpoint-2600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f3606bff6312b72b3b51fbabcad96e085e08d05ec7ead01d6ccd499521240e71 +size 1465 diff --git a/checkpoint-2600/tokenizer.json b/checkpoint-2600/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-2600/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-2600/tokenizer_config.json b/checkpoint-2600/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-2600/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-2600/trainer_state.json b/checkpoint-2600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..5684335c2435566712ce181a5c32cef8624a5bf9 --- /dev/null +++ b/checkpoint-2600/trainer_state.json @@ -0,0 +1,944 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.6971210242316587, + "eval_steps": 500, + "global_step": 2600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.197326861503836e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2600/training_args.bin b/checkpoint-2600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-2600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-2800/README.md b/checkpoint-2800/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-2800/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-2800/adapter_config.json b/checkpoint-2800/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-2800/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-2800/adapter_model.safetensors b/checkpoint-2800/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4237a2b77542cf2b1905c960c34bceae1ef6a0e4 --- /dev/null +++ b/checkpoint-2800/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9341c2981beb59a12df8327cd90b3489aa5a1c57abdae2072b03804ab838b5e6 +size 50360752 diff --git a/checkpoint-2800/chat_template.jinja b/checkpoint-2800/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-2800/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-2800/optimizer.pt b/checkpoint-2800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..dbf5f4fec877e7d07c12999b88b1e371b9d44d78 --- /dev/null +++ b/checkpoint-2800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:02fc4d956c8fbe3378240b6790ff74c8e0253a10478d1fbaeba92d4ec870f7fd +size 100828235 diff --git a/checkpoint-2800/rng_state.pth b/checkpoint-2800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..f68aafde8aac62ace2f4d7890d7bdfb76782b4e0 --- /dev/null +++ b/checkpoint-2800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d5f657d76b555833134e4fa8034da77eac4639aecb8f189e14cd3669258bc7c8 +size 14645 diff --git a/checkpoint-2800/scheduler.pt b/checkpoint-2800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..a1cf24b8870a2ddd09c47e9e33d1efa1fe4a314e --- /dev/null +++ b/checkpoint-2800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d2fd4f05577235783386ce72862134bcceb579f6782c2b37884bd0c38983b508 +size 1465 diff --git a/checkpoint-2800/tokenizer.json b/checkpoint-2800/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-2800/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-2800/tokenizer_config.json b/checkpoint-2800/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-2800/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-2800/trainer_state.json b/checkpoint-2800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..70bed6ffb1cbef27a91668c4a5bbc136cb5eb658 --- /dev/null +++ b/checkpoint-2800/trainer_state.json @@ -0,0 +1,1014 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.7507457184033247, + "eval_steps": 500, + "global_step": 2800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.4429710429456384e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-2800/training_args.bin b/checkpoint-2800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-2800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-3000/README.md b/checkpoint-3000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-3000/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-3000/adapter_config.json b/checkpoint-3000/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-3000/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-3000/adapter_model.safetensors b/checkpoint-3000/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..dc6e448b857b020f1bed075d512499b096ec0243 --- /dev/null +++ b/checkpoint-3000/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:502d5f8ea7c93de3267c6c653b1f66decc4973a650ea547afb6815aa59e41c00 +size 50360752 diff --git a/checkpoint-3000/chat_template.jinja b/checkpoint-3000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-3000/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-3000/optimizer.pt b/checkpoint-3000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a1b96a854f7716fe38d440feb2c6a459a76a15be --- /dev/null +++ b/checkpoint-3000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1699182c47e8d73978a11f7054ce75e1f2000cd96529a49467acad54b95ad137 +size 100828235 diff --git a/checkpoint-3000/rng_state.pth b/checkpoint-3000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..ffd5db9870a47a2e580551277cf767721b3c7a60 --- /dev/null +++ b/checkpoint-3000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5d51c550d5e4ed5ea74f62db61d558c3947bf0e7be99d1f8aa1641764a2899d9 +size 14645 diff --git a/checkpoint-3000/scheduler.pt b/checkpoint-3000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..acac253e62277486eb722d8fa0fe59424e49a316 --- /dev/null +++ b/checkpoint-3000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c82aebb2adaf30bcd259ae87af34af15cdbc12ff481b004f2a325924ae650bc2 +size 1465 diff --git a/checkpoint-3000/tokenizer.json b/checkpoint-3000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-3000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-3000/tokenizer_config.json b/checkpoint-3000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-3000/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-3000/trainer_state.json b/checkpoint-3000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1c2ba4aedfcce97b7d14ed61549e18d461fef833 --- /dev/null +++ b/checkpoint-3000/trainer_state.json @@ -0,0 +1,1084 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.8043704125749908, + "eval_steps": 500, + "global_step": 3000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.6907443999815885e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3000/training_args.bin b/checkpoint-3000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-3000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-3200/README.md b/checkpoint-3200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-3200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-3200/adapter_config.json b/checkpoint-3200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-3200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-3200/adapter_model.safetensors b/checkpoint-3200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a69e00ef65f9370d127c731e96e1f03c1be3d511 --- /dev/null +++ b/checkpoint-3200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:67aab25c78f3ff33249c13ac8ab9e17170269d9733170476f5b00449192d4df5 +size 50360752 diff --git a/checkpoint-3200/chat_template.jinja b/checkpoint-3200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-3200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-3200/optimizer.pt b/checkpoint-3200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..885fbbfb73db64400cc5439d50df267b61573099 --- /dev/null +++ b/checkpoint-3200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:aaec4ae388a2a8f201cb5850524c13702b2f57da221015fcf162dc54ffb5db5f +size 100828235 diff --git a/checkpoint-3200/rng_state.pth b/checkpoint-3200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..7cecc0ab50e8cec158fa68ca2266b4fe3ff95538 --- /dev/null +++ b/checkpoint-3200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8329ae73ab98313c3f32048b8136bc9c1d53fe02a9d15fe134568aff617eb2ab +size 14645 diff --git a/checkpoint-3200/scheduler.pt b/checkpoint-3200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..334db1664d5ec95c81e46299d343bad4d6056ee7 --- /dev/null +++ b/checkpoint-3200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:03c0e2a76cba07205e0fa06795ebf8be7d64722b16bcfeacafeb1fdc874a2023 +size 1465 diff --git a/checkpoint-3200/tokenizer.json b/checkpoint-3200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-3200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-3200/tokenizer_config.json b/checkpoint-3200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-3200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-3200/trainer_state.json b/checkpoint-3200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a9708a9c6d565378ba5892c80d8a2896595861c5 --- /dev/null +++ b/checkpoint-3200/trainer_state.json @@ -0,0 +1,1154 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.8579951067466568, + "eval_steps": 500, + "global_step": 3200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.938011930771415e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3200/training_args.bin b/checkpoint-3200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-3200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-3400/README.md b/checkpoint-3400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-3400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-3400/adapter_config.json b/checkpoint-3400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-3400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-3400/adapter_model.safetensors b/checkpoint-3400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a47e91eecdc4a973d0b317567c9f32d3dc258ac9 --- /dev/null +++ b/checkpoint-3400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c3ec0fd4398b542e929710aa03ee5201c0a58bb48c61514fbe9f243651bb66eb +size 50360752 diff --git a/checkpoint-3400/chat_template.jinja b/checkpoint-3400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-3400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-3400/optimizer.pt b/checkpoint-3400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f683ff93081c04db5cd29dbd813394ee1eeb3084 --- /dev/null +++ b/checkpoint-3400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:67e45e698965aef6a51a58f0bb82a862bcc1242c2b123ba7238328be4399f0f1 +size 100828235 diff --git a/checkpoint-3400/rng_state.pth b/checkpoint-3400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..cddc21f632fdbacd9e9bc19e13ce72e2a3d1ac75 --- /dev/null +++ b/checkpoint-3400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:09b46f94b951b86129463f17a330ae11cf0c1eae5abf93203d5b896e6e628d4f +size 14645 diff --git a/checkpoint-3400/scheduler.pt b/checkpoint-3400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..63c37ca3ad171bf5a6ad23c0273dee7aac46918c --- /dev/null +++ b/checkpoint-3400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a290cdb73f982dc9af70b364cd219af8450564e4a10c0cd208c0c9d4bd2215c0 +size 1465 diff --git a/checkpoint-3400/tokenizer.json b/checkpoint-3400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-3400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-3400/tokenizer_config.json b/checkpoint-3400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-3400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-3400/trainer_state.json b/checkpoint-3400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..3aa05e25dc1f3b976cca01136110820fa99b7b4e --- /dev/null +++ b/checkpoint-3400/trainer_state.json @@ -0,0 +1,1224 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.9116198009183228, + "eval_steps": 500, + "global_step": 3400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.185686979384115e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3400/training_args.bin b/checkpoint-3400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-3400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-3600/README.md b/checkpoint-3600/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-3600/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-3600/adapter_config.json b/checkpoint-3600/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-3600/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-3600/adapter_model.safetensors b/checkpoint-3600/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6398b3c31d809f747a915a887f8f582249aa209c --- /dev/null +++ b/checkpoint-3600/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fee7278ec9639f4bb3f7192769eadf94307962b5bfa164f7ce0398f2f7b1fbdc +size 50360752 diff --git a/checkpoint-3600/chat_template.jinja b/checkpoint-3600/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-3600/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-3600/optimizer.pt b/checkpoint-3600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..cf6731ef0c0288544ead46febe9f6ae45ccdb73f --- /dev/null +++ b/checkpoint-3600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:40c884e18d08a65d9f227d635573d0bae9ddceb241bd4dcb037e094c42034863 +size 100828235 diff --git a/checkpoint-3600/rng_state.pth b/checkpoint-3600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..71b7ff094fad921a4ad832309245545666aca960 --- /dev/null +++ b/checkpoint-3600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a8829d93dfacc727ad08419cac075580443395727c7aeb155053bf5133bb6ead +size 14645 diff --git a/checkpoint-3600/scheduler.pt b/checkpoint-3600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..e636f3df6836282e54c8f286e5395906d5ae6b6e --- /dev/null +++ b/checkpoint-3600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:61368de55afbac7eb28a60f8c94942258b7873b811ea6e873d61969c349866d0 +size 1465 diff --git a/checkpoint-3600/tokenizer.json b/checkpoint-3600/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-3600/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-3600/tokenizer_config.json b/checkpoint-3600/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-3600/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-3600/trainer_state.json b/checkpoint-3600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0d75dc2ac528f07d8ca5a3121f82208d79bad378 --- /dev/null +++ b/checkpoint-3600/trainer_state.json @@ -0,0 +1,1294 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.9652444950899889, + "eval_steps": 500, + "global_step": 3600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.4337636641191526e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3600/training_args.bin b/checkpoint-3600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-3600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-3800/README.md b/checkpoint-3800/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-3800/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-3800/adapter_config.json b/checkpoint-3800/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-3800/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-3800/adapter_model.safetensors b/checkpoint-3800/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..af78446644bf6ef597d8d4875a8d9ddda753f59c --- /dev/null +++ b/checkpoint-3800/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a47dc8a237bb61b99d278738597538aa15c7ddadf8e039f10ae9b5e8b368e3b5 +size 50360752 diff --git a/checkpoint-3800/chat_template.jinja b/checkpoint-3800/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-3800/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-3800/optimizer.pt b/checkpoint-3800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d12ff8aaa7a84f7433a99f0eea7d0c332c2e1f61 --- /dev/null +++ b/checkpoint-3800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7c32f7e4ae37b7239e5cb7e2a846d23dd7a5456d971b1f112afc6bb6c33382c0 +size 100828235 diff --git a/checkpoint-3800/rng_state.pth b/checkpoint-3800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4d9a3c4151e728d94a04a4bf0f74933e0b40801b --- /dev/null +++ b/checkpoint-3800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8035a28d94e2861444b60e73202ade9ad8d6b5e47e931e266011fcb185ffccda +size 14645 diff --git a/checkpoint-3800/scheduler.pt b/checkpoint-3800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..05ac5dcdb42f40390901372bbc1024100a539c6c --- /dev/null +++ b/checkpoint-3800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d0d71bc291351c30a9b43de3f877c602b9454601b2d8ae2d40b580fad0c372ec +size 1465 diff --git a/checkpoint-3800/tokenizer.json b/checkpoint-3800/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-3800/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-3800/tokenizer_config.json b/checkpoint-3800/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-3800/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-3800/trainer_state.json b/checkpoint-3800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..352c401fd44dd4ca31d0a528c512cb989b83c6d0 --- /dev/null +++ b/checkpoint-3800/trainer_state.json @@ -0,0 +1,1364 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.018768642960083, + "eval_steps": 500, + "global_step": 3800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.6827452904938496e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-3800/training_args.bin b/checkpoint-3800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-3800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-400/README.md b/checkpoint-400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-400/adapter_config.json b/checkpoint-400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-400/adapter_model.safetensors b/checkpoint-400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..41ab37938d8249ea07308ff318df96b491e75b4c --- /dev/null +++ b/checkpoint-400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc1f8c7a93a803cf603c10f53f449d2defafb9920543481c424e82b6afb4c7d6 +size 50360752 diff --git a/checkpoint-400/chat_template.jinja b/checkpoint-400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-400/optimizer.pt b/checkpoint-400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..8d96db457b0917b58faa719dae0868ad7e67f55c --- /dev/null +++ b/checkpoint-400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1a793075733206458d44e62e354e6175a015e55dfa0f8ac76da92f1c47e2dbc0 +size 100828235 diff --git a/checkpoint-400/rng_state.pth b/checkpoint-400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..8fd56cdfd0ceef28c611428fff47f7e2e6a93ac6 --- /dev/null +++ b/checkpoint-400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d4d66c1b8ffb7bff6a5dbad802a36584713b4acb078c71e97b0d664ee0cea1ec +size 14645 diff --git a/checkpoint-400/scheduler.pt b/checkpoint-400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..f69bd703793daca9575464b2aef6843304855cfd --- /dev/null +++ b/checkpoint-400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5105cfc7d8f214e6052469b04036343fd59f9ed025860545346c8ee4bab7ffbb +size 1465 diff --git a/checkpoint-400/tokenizer.json b/checkpoint-400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-400/tokenizer_config.json b/checkpoint-400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-400/trainer_state.json b/checkpoint-400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..cc0ff0dd56703e9b4783585fdb2fb08369efe9bb --- /dev/null +++ b/checkpoint-400/trainer_state.json @@ -0,0 +1,174 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.1072493883433321, + "eval_steps": 500, + "global_step": 400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.898960803423642e+16, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-400/training_args.bin b/checkpoint-400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-4000/README.md b/checkpoint-4000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-4000/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-4000/adapter_config.json b/checkpoint-4000/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-4000/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-4000/adapter_model.safetensors b/checkpoint-4000/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..c2f292866793d527ea4da26c15700a0567a0b036 --- /dev/null +++ b/checkpoint-4000/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1753129c326465b61b9f2b91512603e563347c696c7470840e9383d276ed1354 +size 50360752 diff --git a/checkpoint-4000/chat_template.jinja b/checkpoint-4000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-4000/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-4000/optimizer.pt b/checkpoint-4000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..40114c754025380955dec8320709ae225624f2c2 --- /dev/null +++ b/checkpoint-4000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ba9d6810641b1594a7f3512ff5b36560c25767c454d69e27ad89d0424ffe09f4 +size 100828235 diff --git a/checkpoint-4000/rng_state.pth b/checkpoint-4000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..541abbdf017f3e2224388fb36a6a5483c3b53659 --- /dev/null +++ b/checkpoint-4000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e2c135a732e64620e2c2d8601198a5da6963f58eea6921151b2d3a78fc57af9c +size 14645 diff --git a/checkpoint-4000/scheduler.pt b/checkpoint-4000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..e2db81d47a471a7184d394eb56ed90d0f9fc1930 --- /dev/null +++ b/checkpoint-4000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:675f310dd1e4b16f9399e9e931f24613c0f68080a472c0dc6e53f616e9348975 +size 1465 diff --git a/checkpoint-4000/tokenizer.json b/checkpoint-4000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-4000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-4000/tokenizer_config.json b/checkpoint-4000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-4000/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-4000/trainer_state.json b/checkpoint-4000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..550195b1256ad51dd413a0a7b14045aaa85c6648 --- /dev/null +++ b/checkpoint-4000/trainer_state.json @@ -0,0 +1,1434 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0723933371317491, + "eval_steps": 500, + "global_step": 4000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.92759796309205e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-4000/training_args.bin b/checkpoint-4000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-4000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-4200/README.md b/checkpoint-4200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-4200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-4200/adapter_config.json b/checkpoint-4200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-4200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-4200/adapter_model.safetensors b/checkpoint-4200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9c2876da9e4a78b08a1b3fb9c72fc69dc0ffe627 --- /dev/null +++ b/checkpoint-4200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1218b5a13ffd881be6928bc76e8e1f73bea0e216013407f9f975c7b4070cdc95 +size 50360752 diff --git a/checkpoint-4200/chat_template.jinja b/checkpoint-4200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-4200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-4200/optimizer.pt b/checkpoint-4200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f4e46f559904761f3b489621ffcc439aca1cb140 --- /dev/null +++ b/checkpoint-4200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b08a9710b97c94c55a68cb4cc4cecb3cb94b0ac69d2b39f2bb0a34c3a6af83d3 +size 100828235 diff --git a/checkpoint-4200/rng_state.pth b/checkpoint-4200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..fe6f5e3d5dafb37629d3b895c26352eaa5e69572 --- /dev/null +++ b/checkpoint-4200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e8cd3600f35fd154e02403d9b006d3301abebfea18794563b39b24b771c5cb36 +size 14645 diff --git a/checkpoint-4200/scheduler.pt b/checkpoint-4200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5cb5adc051524746891ca5dbcc76bd65e7da7a83 --- /dev/null +++ b/checkpoint-4200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b719e1a8e991070bbe2aaa2873e36bc1c032a531a35c0a22d83ba6170bea31ef +size 1465 diff --git a/checkpoint-4200/tokenizer.json b/checkpoint-4200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-4200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-4200/tokenizer_config.json b/checkpoint-4200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-4200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-4200/trainer_state.json b/checkpoint-4200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f250b790a39dcd909519a3371f1a95c7cfe24baa --- /dev/null +++ b/checkpoint-4200/trainer_state.json @@ -0,0 +1,1504 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.1260180313034152, + "eval_steps": 500, + "global_step": 4200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.176267859338322e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-4200/training_args.bin b/checkpoint-4200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-4200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-4400/README.md b/checkpoint-4400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-4400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-4400/adapter_config.json b/checkpoint-4400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-4400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-4400/adapter_model.safetensors b/checkpoint-4400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d184d7f7c1dec481dc035df11399a7c7cf864489 --- /dev/null +++ b/checkpoint-4400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5825dc7f3591648e0cf7c4ff0842260cc574133c333db11e4899294019d8e09e +size 50360752 diff --git a/checkpoint-4400/chat_template.jinja b/checkpoint-4400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-4400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-4400/optimizer.pt b/checkpoint-4400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..2e51386cca4a04bc34f24e31d2dbfcd44b5379e5 --- /dev/null +++ b/checkpoint-4400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2756659988e3b82b58e6cfeac4948da01697beee526da77cc0e8db9d1f1cc18d +size 100828235 diff --git a/checkpoint-4400/rng_state.pth b/checkpoint-4400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..65a800071b3540e6d69cd8b21d9f47d109be820f --- /dev/null +++ b/checkpoint-4400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15298ff903da8de8196d202d25f163414c9c0666065ac1b553486a4e00a5c028 +size 14645 diff --git a/checkpoint-4400/scheduler.pt b/checkpoint-4400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..9cd48156ef9bdbfd2f9abb3f2f4c167d88a561ae --- /dev/null +++ b/checkpoint-4400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:12b034b4b0107f8077e82d6949164cc02311d2cc9a4806d94aee238c7cb72f6e +size 1465 diff --git a/checkpoint-4400/tokenizer.json b/checkpoint-4400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-4400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-4400/tokenizer_config.json b/checkpoint-4400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-4400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-4400/trainer_state.json b/checkpoint-4400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c29080eddb98d20bc4e59127ddce73f237d281fe --- /dev/null +++ b/checkpoint-4400/trainer_state.json @@ -0,0 +1,1574 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.1796427254750812, + "eval_steps": 500, + "global_step": 4400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.42165408619946e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-4400/training_args.bin b/checkpoint-4400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-4400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-4600/README.md b/checkpoint-4600/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-4600/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-4600/adapter_config.json b/checkpoint-4600/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-4600/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-4600/adapter_model.safetensors b/checkpoint-4600/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4988d25e0054a39486e5f6c97dfe1342d3d8317e --- /dev/null +++ b/checkpoint-4600/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21717bae5311060418f1c58765d4605b8c8ab67a5c650469322feff8f436eddc +size 50360752 diff --git a/checkpoint-4600/chat_template.jinja b/checkpoint-4600/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-4600/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-4600/optimizer.pt b/checkpoint-4600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..efe67fdf9fd13acfaad8ddd1a472042bbaa6b6a3 --- /dev/null +++ b/checkpoint-4600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:01a56e63c21a13dd2832bdd26f3995ece9939bd91ab4bd5966fd458a6834f1a5 +size 100828235 diff --git a/checkpoint-4600/rng_state.pth b/checkpoint-4600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..7385973febfa7df5313c696b7dbf0b9e4bd8ba8f --- /dev/null +++ b/checkpoint-4600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93f56a983a825a4d012386c1eb49c8b7340c83bb0e7b0250c508ac173d9c8d1e +size 14645 diff --git a/checkpoint-4600/scheduler.pt b/checkpoint-4600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ecf27ac3eaf20a282860dd567a6dc7658610106b --- /dev/null +++ b/checkpoint-4600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2e54391a300f76ab81ca47c9b5dc0fa1d5e3a764ba448cc3994ba625dd036edc +size 1465 diff --git a/checkpoint-4600/tokenizer.json b/checkpoint-4600/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-4600/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-4600/tokenizer_config.json b/checkpoint-4600/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-4600/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-4600/trainer_state.json b/checkpoint-4600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..3a2759a7c48e2532e2de5623ea1dbbd6bbabf9db --- /dev/null +++ b/checkpoint-4600/trainer_state.json @@ -0,0 +1,1644 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.2332674196467472, + "eval_steps": 500, + "global_step": 4600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.6720162317143245e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-4600/training_args.bin b/checkpoint-4600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-4600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-4800/README.md b/checkpoint-4800/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-4800/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-4800/adapter_config.json b/checkpoint-4800/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-4800/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-4800/adapter_model.safetensors b/checkpoint-4800/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4c6b8ee9313570c4d510898f3e7ec7e7df2572e2 --- /dev/null +++ b/checkpoint-4800/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d9513a26ca3a56a159421394d3c0b2f38b7abbaf02ac994237dad40f0771795e +size 50360752 diff --git a/checkpoint-4800/chat_template.jinja b/checkpoint-4800/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-4800/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-4800/optimizer.pt b/checkpoint-4800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..eab264233d70f1f033b283ba6d37afb312b04219 --- /dev/null +++ b/checkpoint-4800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:960b7f6c6cd995ac69141e48c170ed41272c7a3dce50a75ea8ec64d985e90bb9 +size 100828235 diff --git a/checkpoint-4800/rng_state.pth b/checkpoint-4800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..26680a6db664a93f70bb4f430d5c58e8e318b85c --- /dev/null +++ b/checkpoint-4800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d8812583303a1ffd00675d7bed0ba594fe1235211733de354da4adbe001aa26 +size 14645 diff --git a/checkpoint-4800/scheduler.pt b/checkpoint-4800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..221678c25457523727c5029ee0b04a7b46300d52 --- /dev/null +++ b/checkpoint-4800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f05c4d2a4a26723c2086c10e751650c4a1e3b250d170a087405862de4b2116f1 +size 1465 diff --git a/checkpoint-4800/tokenizer.json b/checkpoint-4800/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-4800/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-4800/tokenizer_config.json b/checkpoint-4800/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-4800/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-4800/trainer_state.json b/checkpoint-4800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..01c6cf43b446916300a5ef3c554419ec8700794e --- /dev/null +++ b/checkpoint-4800/trainer_state.json @@ -0,0 +1,1714 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.2868921138184133, + "eval_steps": 500, + "global_step": 4800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.917521773072056e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-4800/training_args.bin b/checkpoint-4800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-4800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-5000/README.md b/checkpoint-5000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-5000/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-5000/adapter_config.json b/checkpoint-5000/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-5000/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-5000/adapter_model.safetensors b/checkpoint-5000/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..034de777b048107239afb5af185aae9eecc735e2 --- /dev/null +++ b/checkpoint-5000/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:03a2a5c7ca7066503fa16253b698224bb6f1203f6ca8894cd8fb343d11b788e3 +size 50360752 diff --git a/checkpoint-5000/chat_template.jinja b/checkpoint-5000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-5000/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-5000/optimizer.pt b/checkpoint-5000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..70817699ea05b83b2189e047b2cd061bd86f1850 --- /dev/null +++ b/checkpoint-5000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5334c60b6c07b131dd4b79a5db0b6852a3cdd7c9ca8a02d059967a8a4e2efdf9 +size 100828235 diff --git a/checkpoint-5000/rng_state.pth b/checkpoint-5000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..6a292701c0e5e8c7bf505abb23c59ce1e2f3efbe --- /dev/null +++ b/checkpoint-5000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:660f06b1a6ade2b46bec0a145e16ffee761d68f863ea390644961bd21c7e83dd +size 14645 diff --git a/checkpoint-5000/scheduler.pt b/checkpoint-5000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..9010b7e60015c12ec22b08e77c1ca8781502c2a9 --- /dev/null +++ b/checkpoint-5000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d75f431e14a5ca6bd149963d8053337b6fecbfb6654d3108b1accb7e1005dc39 +size 1465 diff --git a/checkpoint-5000/tokenizer.json b/checkpoint-5000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-5000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-5000/tokenizer_config.json b/checkpoint-5000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-5000/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-5000/trainer_state.json b/checkpoint-5000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f78b35c5e703a5c9f4a3f9b87f42c57f37f6bfb4 --- /dev/null +++ b/checkpoint-5000/trainer_state.json @@ -0,0 +1,1784 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.3405168079900793, + "eval_steps": 500, + "global_step": 5000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.159390743012475e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-5000/training_args.bin b/checkpoint-5000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-5000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-5200/README.md b/checkpoint-5200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-5200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-5200/adapter_config.json b/checkpoint-5200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-5200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-5200/adapter_model.safetensors b/checkpoint-5200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..35f592962753bc8966d03b6525c6a7744a9dec87 --- /dev/null +++ b/checkpoint-5200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:146eef7222fc9494a8bc761b535e0037457f0a128ba28be257ebf7b38dc38ff5 +size 50360752 diff --git a/checkpoint-5200/chat_template.jinja b/checkpoint-5200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-5200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-5200/optimizer.pt b/checkpoint-5200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..2372dd8ef1822cee7c2b846881b93d5ff2dccc96 --- /dev/null +++ b/checkpoint-5200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de3b4bdc043543563bce989493312b7726f9e7e060546c4cb56c159edeab589b +size 100828235 diff --git a/checkpoint-5200/rng_state.pth b/checkpoint-5200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..f3e67e6ade4cb19889ae474828386f05acfbef58 --- /dev/null +++ b/checkpoint-5200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:632c73c4e7c60774a62a54d09cb677cc6262cb44d7e9b66d5cf59806414ad8cc +size 14645 diff --git a/checkpoint-5200/scheduler.pt b/checkpoint-5200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d452350fcd1303c4b911fe507a0241a6e3d842c7 --- /dev/null +++ b/checkpoint-5200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:689a975460e3fd895001e99f36632cb4af3512fd358dd1d8ee5e76dbdbaca867 +size 1465 diff --git a/checkpoint-5200/tokenizer.json b/checkpoint-5200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-5200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-5200/tokenizer_config.json b/checkpoint-5200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-5200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-5200/trainer_state.json b/checkpoint-5200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c5c73620d2deffad18ef3f05179b6ae0376576e4 --- /dev/null +++ b/checkpoint-5200/trainer_state.json @@ -0,0 +1,1854 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.3941415021617454, + "eval_steps": 500, + "global_step": 5200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.407414492442685e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-5200/training_args.bin b/checkpoint-5200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-5200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-5400/README.md b/checkpoint-5400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-5400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-5400/adapter_config.json b/checkpoint-5400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-5400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-5400/adapter_model.safetensors b/checkpoint-5400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2eb4981d81b334c1b3f22817b2ed85fa808e40c8 --- /dev/null +++ b/checkpoint-5400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:28141b68025bab841580b56ce0468675165e7f78435de7dc1cb94d15a3148cd7 +size 50360752 diff --git a/checkpoint-5400/chat_template.jinja b/checkpoint-5400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-5400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-5400/optimizer.pt b/checkpoint-5400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..3777ddaadc140a28e736e2477114b47abc2661a5 --- /dev/null +++ b/checkpoint-5400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0155b3df218cbedbd2627d9c8b99b5b0c375c119c891304d6db43641e80a0281 +size 100828235 diff --git a/checkpoint-5400/rng_state.pth b/checkpoint-5400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..77ac9b767ad0fb17455128c37a9bc3ef30efc8bf --- /dev/null +++ b/checkpoint-5400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:defa5f944fd5a2264e613ebd98def7c3d56b4346aef74846345ce7b7eb28580a +size 14645 diff --git a/checkpoint-5400/scheduler.pt b/checkpoint-5400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..9f5c790d78e0557fe9d583379cb31e3c56c6049d --- /dev/null +++ b/checkpoint-5400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:adf338bfedeceeaa4c29010d52f250a866fed7a9ac20388b9a7570b82a22b7d3 +size 1465 diff --git a/checkpoint-5400/tokenizer.json b/checkpoint-5400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-5400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-5400/tokenizer_config.json b/checkpoint-5400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-5400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-5400/trainer_state.json b/checkpoint-5400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..5595a1d764c05d693da620a109e276392e58bdd7 --- /dev/null +++ b/checkpoint-5400/trainer_state.json @@ -0,0 +1,1924 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.4477661963334114, + "eval_steps": 500, + "global_step": 5400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.652352029577196e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-5400/training_args.bin b/checkpoint-5400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-5400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-5600/README.md b/checkpoint-5600/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-5600/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-5600/adapter_config.json b/checkpoint-5600/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-5600/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-5600/adapter_model.safetensors b/checkpoint-5600/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6da8258121afcff7481ebd61dbc536630a39b97d --- /dev/null +++ b/checkpoint-5600/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3ea4e2bcd387e044005e814063b1225dbfc8890aa509e03eb1431bf2d3644ea9 +size 50360752 diff --git a/checkpoint-5600/chat_template.jinja b/checkpoint-5600/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-5600/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-5600/optimizer.pt b/checkpoint-5600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6db6396492016f102fa451cebb4b171354daad8c --- /dev/null +++ b/checkpoint-5600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be14ed84978828c6cc9295376bb460c350d5ac0806608f3458fa1c7a039fdee0 +size 100828235 diff --git a/checkpoint-5600/rng_state.pth b/checkpoint-5600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4368707c059765bc10ef490ba129ad99bb506e57 --- /dev/null +++ b/checkpoint-5600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5791c2a0d1c09d1732c89ce549a6e832485b37524facaa2642c76169e19f6d1b +size 14645 diff --git a/checkpoint-5600/scheduler.pt b/checkpoint-5600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..f980c2969f5c2b349c98715d6aa548a3fd19dea8 --- /dev/null +++ b/checkpoint-5600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:56f6cace1076b12bf8584bd4d77224683396a0269b67fae379ebc20bbe096585 +size 1465 diff --git a/checkpoint-5600/tokenizer.json b/checkpoint-5600/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-5600/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-5600/tokenizer_config.json b/checkpoint-5600/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-5600/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-5600/trainer_state.json b/checkpoint-5600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d8f8c3643238fd2be19c40f080d5cb354693d286 --- /dev/null +++ b/checkpoint-5600/trainer_state.json @@ -0,0 +1,1994 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.5013908905050775, + "eval_steps": 500, + "global_step": 5600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.895036875406295e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-5600/training_args.bin b/checkpoint-5600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-5600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-5800/README.md b/checkpoint-5800/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-5800/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-5800/adapter_config.json b/checkpoint-5800/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-5800/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-5800/adapter_model.safetensors b/checkpoint-5800/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a403e364d512b47cd8cae7d992e2ef182d8128d9 --- /dev/null +++ b/checkpoint-5800/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:48a6c661168045f149a8bf9fc70b56e1a98a50603045f9a0b0319be5ef6c3b8a +size 50360752 diff --git a/checkpoint-5800/chat_template.jinja b/checkpoint-5800/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-5800/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-5800/optimizer.pt b/checkpoint-5800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..3579dc7b1d282e75ad6f46fa6b4f37c59ffd469a --- /dev/null +++ b/checkpoint-5800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cf8b53f5878689f0298adf6edbc474223bdf44d344afd820c162f6f48104bc98 +size 100828235 diff --git a/checkpoint-5800/rng_state.pth b/checkpoint-5800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..528271c23234b9bc5137a624b9af989c64064949 --- /dev/null +++ b/checkpoint-5800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b98fa80b2bfcc9eb38e63a48bdf8e54218d3a8d41b0da0c5eaf3276e952ed7f5 +size 14645 diff --git a/checkpoint-5800/scheduler.pt b/checkpoint-5800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..73c36f4c111d4a502f43f67a4f83e0c83fa011cb --- /dev/null +++ b/checkpoint-5800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ceac609cbebd0e97bb85a708c488e23e04b9da9d0b459db394ef36def3d7040d +size 1465 diff --git a/checkpoint-5800/tokenizer.json b/checkpoint-5800/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-5800/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-5800/tokenizer_config.json b/checkpoint-5800/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-5800/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-5800/trainer_state.json b/checkpoint-5800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..2d285985ba9e8a933277746d56d395f1075304cd --- /dev/null +++ b/checkpoint-5800/trainer_state.json @@ -0,0 +1,2064 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.5550155846767435, + "eval_steps": 500, + "global_step": 5800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.141796059221197e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-5800/training_args.bin b/checkpoint-5800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-5800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-600/README.md b/checkpoint-600/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-600/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-600/adapter_config.json b/checkpoint-600/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-600/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-600/adapter_model.safetensors b/checkpoint-600/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..1b0beae437d9f3b7de85a27efcbbe36851e34a54 --- /dev/null +++ b/checkpoint-600/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f1169af160e019a7373844aea7c18a0972e25fcc00f8db1ca2d7b37c1fc53cfa +size 50360752 diff --git a/checkpoint-600/chat_template.jinja b/checkpoint-600/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-600/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-600/optimizer.pt b/checkpoint-600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..41ac24db16c954b7c3a4f6b6d9025ddb075e428d --- /dev/null +++ b/checkpoint-600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f8e9ba6d8858deca4693da4556589ab009b0c0acd0f5709f9eea9cf3ddc048e +size 100828235 diff --git a/checkpoint-600/rng_state.pth b/checkpoint-600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9bb823fba828655ac661a72989f8259de457c6a3 --- /dev/null +++ b/checkpoint-600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c8f869532707a1216d0a788bfd4dc1d65138d6bd118120f551cacd491bbb96b4 +size 14645 diff --git a/checkpoint-600/scheduler.pt b/checkpoint-600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d83fbc73c447eba14394d91a02cba49bb3135f12 --- /dev/null +++ b/checkpoint-600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21c2421bb1edc243038582e86c94a26d67145168f5438f0ca7b23a942e7dfa1e +size 1465 diff --git a/checkpoint-600/tokenizer.json b/checkpoint-600/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-600/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-600/tokenizer_config.json b/checkpoint-600/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-600/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-600/trainer_state.json b/checkpoint-600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..8744fac2b14347b1e915819a2dc59f9f44938ff9 --- /dev/null +++ b/checkpoint-600/trainer_state.json @@ -0,0 +1,244 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.16087408251499816, + "eval_steps": 500, + "global_step": 600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.35254579186688e+16, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-600/training_args.bin b/checkpoint-600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-6000/README.md b/checkpoint-6000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-6000/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-6000/adapter_config.json b/checkpoint-6000/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-6000/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-6000/adapter_model.safetensors b/checkpoint-6000/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3e8405737453d00c7da7cfa33ffa3072b1f74072 --- /dev/null +++ b/checkpoint-6000/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1d7da8f2e0bd7e4fbddb5eb821bd5ba03f35be2409a0e5bdd187c58c86db0b05 +size 50360752 diff --git a/checkpoint-6000/chat_template.jinja b/checkpoint-6000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-6000/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-6000/optimizer.pt b/checkpoint-6000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..4765cada5d928d31a3664daf96a0eba8dc39d857 --- /dev/null +++ b/checkpoint-6000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:86954006a24eeedce074114af943943910cd56e82b6aa5b8f1e08768cbd71242 +size 100828235 diff --git a/checkpoint-6000/rng_state.pth b/checkpoint-6000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..1b02f4f798a6e4e3655dc8af821c53672ef6a9cf --- /dev/null +++ b/checkpoint-6000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d761063cf2e98a79d7f1635899deea6cdb52acd8964f54fe373e307bbf2cbad1 +size 14645 diff --git a/checkpoint-6000/scheduler.pt b/checkpoint-6000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..f1a9fd4e828798e91c05aa7abf28c77d75b8bc33 --- /dev/null +++ b/checkpoint-6000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6eb48adcee1fa0c3d7fb5270844c86c911e478414ee44487aa4d5df5179e49f9 +size 1465 diff --git a/checkpoint-6000/tokenizer.json b/checkpoint-6000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-6000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-6000/tokenizer_config.json b/checkpoint-6000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-6000/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-6000/trainer_state.json b/checkpoint-6000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..daa9046fb9562217c3a0cbd6a52feb9c315f611c --- /dev/null +++ b/checkpoint-6000/trainer_state.json @@ -0,0 +1,2134 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.6086402788484095, + "eval_steps": 500, + "global_step": 6000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.388858570735186e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-6000/training_args.bin b/checkpoint-6000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-6000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-6200/README.md b/checkpoint-6200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-6200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-6200/adapter_config.json b/checkpoint-6200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-6200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-6200/adapter_model.safetensors b/checkpoint-6200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..67f488e84b33f07e28f288afaea202eda2de8ff6 --- /dev/null +++ b/checkpoint-6200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ecddfa1ad15326aa3aad668ace2624b7406005818d1ccc033c855dd65daa0af1 +size 50360752 diff --git a/checkpoint-6200/chat_template.jinja b/checkpoint-6200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-6200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-6200/optimizer.pt b/checkpoint-6200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..3b6371526b888677e438cb71d951d9003aee3ec6 --- /dev/null +++ b/checkpoint-6200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:063c20a9fc6d1290e39711ff8109126773c0d1e8e51e63de2ba6d87f78f40c1f +size 100828235 diff --git a/checkpoint-6200/rng_state.pth b/checkpoint-6200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..181e79122595a400a196a12bc39acf21da26f8ba --- /dev/null +++ b/checkpoint-6200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:330ee16477886512f1c2dcb9d2dd56ad6f274db1cf0feb2197d89de7803d2c2a +size 14645 diff --git a/checkpoint-6200/scheduler.pt b/checkpoint-6200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..db9e284ba6b2e4d27e044d4fff2c2a7b3182e6da --- /dev/null +++ b/checkpoint-6200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9a33bf06e2136cdac84aaeda236d77d618aa9aabbf0388ae72b32a51bfebf8fa +size 1465 diff --git a/checkpoint-6200/tokenizer.json b/checkpoint-6200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-6200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-6200/tokenizer_config.json b/checkpoint-6200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-6200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-6200/trainer_state.json b/checkpoint-6200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..87d941ff821af8c115342b086348d4cd693a43ee --- /dev/null +++ b/checkpoint-6200/trainer_state.json @@ -0,0 +1,2204 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.6622649730200756, + "eval_steps": 500, + "global_step": 6200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.631987064833311e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-6200/training_args.bin b/checkpoint-6200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-6200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-6400/README.md b/checkpoint-6400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-6400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-6400/adapter_config.json b/checkpoint-6400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-6400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-6400/adapter_model.safetensors b/checkpoint-6400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a63085d98b2fa346a94c8ce501dc1ca67eac67af --- /dev/null +++ b/checkpoint-6400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:449b3b4c9fca8ac1f4f4392005b516f91382fe9d2311f3675a9d1ca8f9081f68 +size 50360752 diff --git a/checkpoint-6400/chat_template.jinja b/checkpoint-6400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-6400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-6400/optimizer.pt b/checkpoint-6400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..601e5f97b282ab0632a0f8498ceff8bce151d044 --- /dev/null +++ b/checkpoint-6400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:abce19518619687e5a1c0090a291dd0ad478c935d4ad81a66becd5c451f30aa5 +size 100828235 diff --git a/checkpoint-6400/rng_state.pth b/checkpoint-6400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..3ee1dd1b014381d33facdb6549fb6b59b2faa38c --- /dev/null +++ b/checkpoint-6400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d90b097f445f1e554b6312c9ebf2c9fa0b31dc51bedadaa8f8fbc4a3864d8a2f +size 14645 diff --git a/checkpoint-6400/scheduler.pt b/checkpoint-6400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..2aa35e9b8f442b2db543507ca4bd17fbb0ae4efc --- /dev/null +++ b/checkpoint-6400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9e2add601d450aaa7de8db6075b3038505d99dd0b0cab0c4e047224526f466c7 +size 1465 diff --git a/checkpoint-6400/tokenizer.json b/checkpoint-6400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-6400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-6400/tokenizer_config.json b/checkpoint-6400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-6400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-6400/trainer_state.json b/checkpoint-6400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f57595567d78d9babdb1a0b3d9d7ea43cb260923 --- /dev/null +++ b/checkpoint-6400/trainer_state.json @@ -0,0 +1,2274 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.7158896671917419, + "eval_steps": 500, + "global_step": 6400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + }, + { + "epoch": 1.6676274424372424, + "grad_norm": 0.22097784280776978, + "learning_rate": 2.4953083109919572e-06, + "loss": 0.43366107940673826, + "step": 6220 + }, + { + "epoch": 1.672989911854409, + "grad_norm": 0.2299281805753708, + "learning_rate": 2.455093833780161e-06, + "loss": 0.5145821094512939, + "step": 6240 + }, + { + "epoch": 1.6783523812715755, + "grad_norm": 0.2384844720363617, + "learning_rate": 2.4148793565683646e-06, + "loss": 0.459308385848999, + "step": 6260 + }, + { + "epoch": 1.6837148506887423, + "grad_norm": 0.24471035599708557, + "learning_rate": 2.3746648793565683e-06, + "loss": 0.4676504611968994, + "step": 6280 + }, + { + "epoch": 1.6890773201059086, + "grad_norm": 0.24419866502285004, + "learning_rate": 2.334450402144772e-06, + "loss": 0.4745138168334961, + "step": 6300 + }, + { + "epoch": 1.6944397895230754, + "grad_norm": 0.15896575152873993, + "learning_rate": 2.294235924932976e-06, + "loss": 0.5073649883270264, + "step": 6320 + }, + { + "epoch": 1.699802258940242, + "grad_norm": 0.26504868268966675, + "learning_rate": 2.25402144772118e-06, + "loss": 0.4534353733062744, + "step": 6340 + }, + { + "epoch": 1.7051647283574085, + "grad_norm": 0.2461850792169571, + "learning_rate": 2.2138069705093836e-06, + "loss": 0.4862947940826416, + "step": 6360 + }, + { + "epoch": 1.7105271977745753, + "grad_norm": 0.17332817614078522, + "learning_rate": 2.1735924932975873e-06, + "loss": 0.5049370765686035, + "step": 6380 + }, + { + "epoch": 1.7158896671917419, + "grad_norm": 0.19762548804283142, + "learning_rate": 2.133378016085791e-06, + "loss": 0.5272616386413574, + "step": 6400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.880667884237722e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-6400/training_args.bin b/checkpoint-6400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-6400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-6600/README.md b/checkpoint-6600/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-6600/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-6600/adapter_config.json b/checkpoint-6600/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-6600/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-6600/adapter_model.safetensors b/checkpoint-6600/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5f09d27c16c9d9057d6be385d0d158b846b7e3e0 --- /dev/null +++ b/checkpoint-6600/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:07858729b9755288cd707add485cd622a969df7813c4e7957784677567fde546 +size 50360752 diff --git a/checkpoint-6600/chat_template.jinja b/checkpoint-6600/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-6600/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-6600/optimizer.pt b/checkpoint-6600/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..52e7c940290592ea4b76ac347c41d62375454d64 --- /dev/null +++ b/checkpoint-6600/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c8e43a0540ac047a9fcc2c58c25d2bb53c1b8a83fc85ceb99d55eebd6e04093 +size 100828235 diff --git a/checkpoint-6600/rng_state.pth b/checkpoint-6600/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..563750c0255da92880e11402fc1b2dadfc0b5d65 --- /dev/null +++ b/checkpoint-6600/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:16843b9ee8bbfd86b883e92d6739e6aa0aa96746771df4a0ef0f3191b38210b7 +size 14645 diff --git a/checkpoint-6600/scheduler.pt b/checkpoint-6600/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..f6a5e4822738abe47a37d12a83cd5178a2bc0574 --- /dev/null +++ b/checkpoint-6600/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:25bfade1230c63a6e64bbdacfdf5f14af936aeb3654bdd194db91fbf8f3883cd +size 1465 diff --git a/checkpoint-6600/tokenizer.json b/checkpoint-6600/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-6600/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-6600/tokenizer_config.json b/checkpoint-6600/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-6600/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-6600/trainer_state.json b/checkpoint-6600/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a6f2244bcc9bba8f26556b3462a1f3d9b10fb0a7 --- /dev/null +++ b/checkpoint-6600/trainer_state.json @@ -0,0 +1,2344 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.769514361363408, + "eval_steps": 500, + "global_step": 6600, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + }, + { + "epoch": 1.6676274424372424, + "grad_norm": 0.22097784280776978, + "learning_rate": 2.4953083109919572e-06, + "loss": 0.43366107940673826, + "step": 6220 + }, + { + "epoch": 1.672989911854409, + "grad_norm": 0.2299281805753708, + "learning_rate": 2.455093833780161e-06, + "loss": 0.5145821094512939, + "step": 6240 + }, + { + "epoch": 1.6783523812715755, + "grad_norm": 0.2384844720363617, + "learning_rate": 2.4148793565683646e-06, + "loss": 0.459308385848999, + "step": 6260 + }, + { + "epoch": 1.6837148506887423, + "grad_norm": 0.24471035599708557, + "learning_rate": 2.3746648793565683e-06, + "loss": 0.4676504611968994, + "step": 6280 + }, + { + "epoch": 1.6890773201059086, + "grad_norm": 0.24419866502285004, + "learning_rate": 2.334450402144772e-06, + "loss": 0.4745138168334961, + "step": 6300 + }, + { + "epoch": 1.6944397895230754, + "grad_norm": 0.15896575152873993, + "learning_rate": 2.294235924932976e-06, + "loss": 0.5073649883270264, + "step": 6320 + }, + { + "epoch": 1.699802258940242, + "grad_norm": 0.26504868268966675, + "learning_rate": 2.25402144772118e-06, + "loss": 0.4534353733062744, + "step": 6340 + }, + { + "epoch": 1.7051647283574085, + "grad_norm": 0.2461850792169571, + "learning_rate": 2.2138069705093836e-06, + "loss": 0.4862947940826416, + "step": 6360 + }, + { + "epoch": 1.7105271977745753, + "grad_norm": 0.17332817614078522, + "learning_rate": 2.1735924932975873e-06, + "loss": 0.5049370765686035, + "step": 6380 + }, + { + "epoch": 1.7158896671917419, + "grad_norm": 0.19762548804283142, + "learning_rate": 2.133378016085791e-06, + "loss": 0.5272616386413574, + "step": 6400 + }, + { + "epoch": 1.7212521366089084, + "grad_norm": 0.23265399038791656, + "learning_rate": 2.0931635388739946e-06, + "loss": 0.47600841522216797, + "step": 6420 + }, + { + "epoch": 1.726614606026075, + "grad_norm": 0.20868578553199768, + "learning_rate": 2.0529490616621983e-06, + "loss": 0.5027226448059082, + "step": 6440 + }, + { + "epoch": 1.7319770754432415, + "grad_norm": 0.2851981520652771, + "learning_rate": 2.012734584450402e-06, + "loss": 0.5288124561309815, + "step": 6460 + }, + { + "epoch": 1.7373395448604083, + "grad_norm": 0.20086587965488434, + "learning_rate": 1.9725201072386057e-06, + "loss": 0.4625516891479492, + "step": 6480 + }, + { + "epoch": 1.7427020142775749, + "grad_norm": 0.24060192704200745, + "learning_rate": 1.93230563002681e-06, + "loss": 0.4843903541564941, + "step": 6500 + }, + { + "epoch": 1.7480644836947414, + "grad_norm": 0.33561915159225464, + "learning_rate": 1.8920911528150133e-06, + "loss": 0.4823720932006836, + "step": 6520 + }, + { + "epoch": 1.753426953111908, + "grad_norm": 0.2510465383529663, + "learning_rate": 1.851876675603217e-06, + "loss": 0.46517143249511717, + "step": 6540 + }, + { + "epoch": 1.7587894225290746, + "grad_norm": 0.2631177604198456, + "learning_rate": 1.811662198391421e-06, + "loss": 0.5004732131958007, + "step": 6560 + }, + { + "epoch": 1.7641518919462413, + "grad_norm": 0.3493230640888214, + "learning_rate": 1.7714477211796249e-06, + "loss": 0.523811674118042, + "step": 6580 + }, + { + "epoch": 1.769514361363408, + "grad_norm": 0.1742691546678543, + "learning_rate": 1.7312332439678286e-06, + "loss": 0.5276295661926269, + "step": 6600 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.127024591687352e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-6600/training_args.bin b/checkpoint-6600/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-6600/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-6800/README.md b/checkpoint-6800/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-6800/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-6800/adapter_config.json b/checkpoint-6800/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-6800/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-6800/adapter_model.safetensors b/checkpoint-6800/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4e5ced89a23e0ee4d928e82115bef05251adaca4 --- /dev/null +++ b/checkpoint-6800/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ed2991897b87a6505a11f1e261458d003853a423454d0c695a836279c91dedb0 +size 50360752 diff --git a/checkpoint-6800/chat_template.jinja b/checkpoint-6800/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-6800/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-6800/optimizer.pt b/checkpoint-6800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6f6ead2116d20e12680868a130aa5f4c5ea5b44a --- /dev/null +++ b/checkpoint-6800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b06c5d417d484fe30a3f5e545864a81855d53a261a19a9221db0ec0ba0ebdf6e +size 100828235 diff --git a/checkpoint-6800/rng_state.pth b/checkpoint-6800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..838e7749ecb01948a0070debc676a56f542d313f --- /dev/null +++ b/checkpoint-6800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:289bb67593ed6c88781bd4c977b0573cb2a3f9659d6a230efdd16085a8520afe +size 14645 diff --git a/checkpoint-6800/scheduler.pt b/checkpoint-6800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..b279a01beb2afedbb029a4aa724d9541f3112902 --- /dev/null +++ b/checkpoint-6800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:88a909cddd57590705e606f63cdba0fe6697d4a9729afb4be3dfced9292a5a67 +size 1465 diff --git a/checkpoint-6800/tokenizer.json b/checkpoint-6800/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-6800/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-6800/tokenizer_config.json b/checkpoint-6800/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-6800/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-6800/trainer_state.json b/checkpoint-6800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9b1cc6607083179f9687b045a191c304affafc26 --- /dev/null +++ b/checkpoint-6800/trainer_state.json @@ -0,0 +1,2414 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.823139055535074, + "eval_steps": 500, + "global_step": 6800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + }, + { + "epoch": 1.6676274424372424, + "grad_norm": 0.22097784280776978, + "learning_rate": 2.4953083109919572e-06, + "loss": 0.43366107940673826, + "step": 6220 + }, + { + "epoch": 1.672989911854409, + "grad_norm": 0.2299281805753708, + "learning_rate": 2.455093833780161e-06, + "loss": 0.5145821094512939, + "step": 6240 + }, + { + "epoch": 1.6783523812715755, + "grad_norm": 0.2384844720363617, + "learning_rate": 2.4148793565683646e-06, + "loss": 0.459308385848999, + "step": 6260 + }, + { + "epoch": 1.6837148506887423, + "grad_norm": 0.24471035599708557, + "learning_rate": 2.3746648793565683e-06, + "loss": 0.4676504611968994, + "step": 6280 + }, + { + "epoch": 1.6890773201059086, + "grad_norm": 0.24419866502285004, + "learning_rate": 2.334450402144772e-06, + "loss": 0.4745138168334961, + "step": 6300 + }, + { + "epoch": 1.6944397895230754, + "grad_norm": 0.15896575152873993, + "learning_rate": 2.294235924932976e-06, + "loss": 0.5073649883270264, + "step": 6320 + }, + { + "epoch": 1.699802258940242, + "grad_norm": 0.26504868268966675, + "learning_rate": 2.25402144772118e-06, + "loss": 0.4534353733062744, + "step": 6340 + }, + { + "epoch": 1.7051647283574085, + "grad_norm": 0.2461850792169571, + "learning_rate": 2.2138069705093836e-06, + "loss": 0.4862947940826416, + "step": 6360 + }, + { + "epoch": 1.7105271977745753, + "grad_norm": 0.17332817614078522, + "learning_rate": 2.1735924932975873e-06, + "loss": 0.5049370765686035, + "step": 6380 + }, + { + "epoch": 1.7158896671917419, + "grad_norm": 0.19762548804283142, + "learning_rate": 2.133378016085791e-06, + "loss": 0.5272616386413574, + "step": 6400 + }, + { + "epoch": 1.7212521366089084, + "grad_norm": 0.23265399038791656, + "learning_rate": 2.0931635388739946e-06, + "loss": 0.47600841522216797, + "step": 6420 + }, + { + "epoch": 1.726614606026075, + "grad_norm": 0.20868578553199768, + "learning_rate": 2.0529490616621983e-06, + "loss": 0.5027226448059082, + "step": 6440 + }, + { + "epoch": 1.7319770754432415, + "grad_norm": 0.2851981520652771, + "learning_rate": 2.012734584450402e-06, + "loss": 0.5288124561309815, + "step": 6460 + }, + { + "epoch": 1.7373395448604083, + "grad_norm": 0.20086587965488434, + "learning_rate": 1.9725201072386057e-06, + "loss": 0.4625516891479492, + "step": 6480 + }, + { + "epoch": 1.7427020142775749, + "grad_norm": 0.24060192704200745, + "learning_rate": 1.93230563002681e-06, + "loss": 0.4843903541564941, + "step": 6500 + }, + { + "epoch": 1.7480644836947414, + "grad_norm": 0.33561915159225464, + "learning_rate": 1.8920911528150133e-06, + "loss": 0.4823720932006836, + "step": 6520 + }, + { + "epoch": 1.753426953111908, + "grad_norm": 0.2510465383529663, + "learning_rate": 1.851876675603217e-06, + "loss": 0.46517143249511717, + "step": 6540 + }, + { + "epoch": 1.7587894225290746, + "grad_norm": 0.2631177604198456, + "learning_rate": 1.811662198391421e-06, + "loss": 0.5004732131958007, + "step": 6560 + }, + { + "epoch": 1.7641518919462413, + "grad_norm": 0.3493230640888214, + "learning_rate": 1.7714477211796249e-06, + "loss": 0.523811674118042, + "step": 6580 + }, + { + "epoch": 1.769514361363408, + "grad_norm": 0.1742691546678543, + "learning_rate": 1.7312332439678286e-06, + "loss": 0.5276295661926269, + "step": 6600 + }, + { + "epoch": 1.7748768307805745, + "grad_norm": 0.16134823858737946, + "learning_rate": 1.6910187667560323e-06, + "loss": 0.5352637290954589, + "step": 6620 + }, + { + "epoch": 1.780239300197741, + "grad_norm": 0.20977018773555756, + "learning_rate": 1.650804289544236e-06, + "loss": 0.4955774784088135, + "step": 6640 + }, + { + "epoch": 1.7856017696149076, + "grad_norm": 0.20511005818843842, + "learning_rate": 1.6105898123324397e-06, + "loss": 0.48643174171447756, + "step": 6660 + }, + { + "epoch": 1.7909642390320744, + "grad_norm": 0.23870044946670532, + "learning_rate": 1.5703753351206434e-06, + "loss": 0.4673162460327148, + "step": 6680 + }, + { + "epoch": 1.796326708449241, + "grad_norm": 0.21660065650939941, + "learning_rate": 1.5301608579088473e-06, + "loss": 0.5381903648376465, + "step": 6700 + }, + { + "epoch": 1.8016891778664075, + "grad_norm": 0.26977139711380005, + "learning_rate": 1.489946380697051e-06, + "loss": 0.42094998359680175, + "step": 6720 + }, + { + "epoch": 1.807051647283574, + "grad_norm": 0.2088550478219986, + "learning_rate": 1.4497319034852549e-06, + "loss": 0.49211792945861815, + "step": 6740 + }, + { + "epoch": 1.8124141167007406, + "grad_norm": 0.18141885101795197, + "learning_rate": 1.4095174262734586e-06, + "loss": 0.46572179794311525, + "step": 6760 + }, + { + "epoch": 1.8177765861179074, + "grad_norm": 0.2200685739517212, + "learning_rate": 1.3693029490616623e-06, + "loss": 0.4996177196502686, + "step": 6780 + }, + { + "epoch": 1.823139055535074, + "grad_norm": 0.19545452296733856, + "learning_rate": 1.329088471849866e-06, + "loss": 0.4731945514678955, + "step": 6800 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.373542625780265e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-6800/training_args.bin b/checkpoint-6800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-6800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-7000/README.md b/checkpoint-7000/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-7000/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-7000/adapter_config.json b/checkpoint-7000/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-7000/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-7000/adapter_model.safetensors b/checkpoint-7000/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5eeaeda2a79ecf256eb58ab7d9efbca0300c5110 --- /dev/null +++ b/checkpoint-7000/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b4663c3b7fb54ea5bdcd3bc60c04176a5052f0c730e5bd2de51576073db6ce7d +size 50360752 diff --git a/checkpoint-7000/chat_template.jinja b/checkpoint-7000/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-7000/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-7000/optimizer.pt b/checkpoint-7000/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..cef40d2996fba6f4adb2c9b2294c1750fb72af35 --- /dev/null +++ b/checkpoint-7000/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d1ef1c8f148b60b0cd7d3d6796071840dec41f4ba0a97495b85fb7c90ffdee79 +size 100828235 diff --git a/checkpoint-7000/rng_state.pth b/checkpoint-7000/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..9dcf635d0af42decbc7ed01d7bf422f16e7c6d6f --- /dev/null +++ b/checkpoint-7000/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:71b1ac65a6bc1245f150fab07c74f73f6f644042151c39b0c3ed7f07bddf91af +size 14645 diff --git a/checkpoint-7000/scheduler.pt b/checkpoint-7000/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..fca355e69583b905245f88112070c3006ad292d2 --- /dev/null +++ b/checkpoint-7000/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:421f24085050d20209d7d03e1cc26d43deaa27183041a8fc1f7d3f63c3e6666c +size 1465 diff --git a/checkpoint-7000/tokenizer.json b/checkpoint-7000/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-7000/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-7000/tokenizer_config.json b/checkpoint-7000/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-7000/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-7000/trainer_state.json b/checkpoint-7000/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..23dfec5dec1b6d1b14c0f9bc5e45c9ee5dc992a0 --- /dev/null +++ b/checkpoint-7000/trainer_state.json @@ -0,0 +1,2484 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.87676374970674, + "eval_steps": 500, + "global_step": 7000, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + }, + { + "epoch": 1.6676274424372424, + "grad_norm": 0.22097784280776978, + "learning_rate": 2.4953083109919572e-06, + "loss": 0.43366107940673826, + "step": 6220 + }, + { + "epoch": 1.672989911854409, + "grad_norm": 0.2299281805753708, + "learning_rate": 2.455093833780161e-06, + "loss": 0.5145821094512939, + "step": 6240 + }, + { + "epoch": 1.6783523812715755, + "grad_norm": 0.2384844720363617, + "learning_rate": 2.4148793565683646e-06, + "loss": 0.459308385848999, + "step": 6260 + }, + { + "epoch": 1.6837148506887423, + "grad_norm": 0.24471035599708557, + "learning_rate": 2.3746648793565683e-06, + "loss": 0.4676504611968994, + "step": 6280 + }, + { + "epoch": 1.6890773201059086, + "grad_norm": 0.24419866502285004, + "learning_rate": 2.334450402144772e-06, + "loss": 0.4745138168334961, + "step": 6300 + }, + { + "epoch": 1.6944397895230754, + "grad_norm": 0.15896575152873993, + "learning_rate": 2.294235924932976e-06, + "loss": 0.5073649883270264, + "step": 6320 + }, + { + "epoch": 1.699802258940242, + "grad_norm": 0.26504868268966675, + "learning_rate": 2.25402144772118e-06, + "loss": 0.4534353733062744, + "step": 6340 + }, + { + "epoch": 1.7051647283574085, + "grad_norm": 0.2461850792169571, + "learning_rate": 2.2138069705093836e-06, + "loss": 0.4862947940826416, + "step": 6360 + }, + { + "epoch": 1.7105271977745753, + "grad_norm": 0.17332817614078522, + "learning_rate": 2.1735924932975873e-06, + "loss": 0.5049370765686035, + "step": 6380 + }, + { + "epoch": 1.7158896671917419, + "grad_norm": 0.19762548804283142, + "learning_rate": 2.133378016085791e-06, + "loss": 0.5272616386413574, + "step": 6400 + }, + { + "epoch": 1.7212521366089084, + "grad_norm": 0.23265399038791656, + "learning_rate": 2.0931635388739946e-06, + "loss": 0.47600841522216797, + "step": 6420 + }, + { + "epoch": 1.726614606026075, + "grad_norm": 0.20868578553199768, + "learning_rate": 2.0529490616621983e-06, + "loss": 0.5027226448059082, + "step": 6440 + }, + { + "epoch": 1.7319770754432415, + "grad_norm": 0.2851981520652771, + "learning_rate": 2.012734584450402e-06, + "loss": 0.5288124561309815, + "step": 6460 + }, + { + "epoch": 1.7373395448604083, + "grad_norm": 0.20086587965488434, + "learning_rate": 1.9725201072386057e-06, + "loss": 0.4625516891479492, + "step": 6480 + }, + { + "epoch": 1.7427020142775749, + "grad_norm": 0.24060192704200745, + "learning_rate": 1.93230563002681e-06, + "loss": 0.4843903541564941, + "step": 6500 + }, + { + "epoch": 1.7480644836947414, + "grad_norm": 0.33561915159225464, + "learning_rate": 1.8920911528150133e-06, + "loss": 0.4823720932006836, + "step": 6520 + }, + { + "epoch": 1.753426953111908, + "grad_norm": 0.2510465383529663, + "learning_rate": 1.851876675603217e-06, + "loss": 0.46517143249511717, + "step": 6540 + }, + { + "epoch": 1.7587894225290746, + "grad_norm": 0.2631177604198456, + "learning_rate": 1.811662198391421e-06, + "loss": 0.5004732131958007, + "step": 6560 + }, + { + "epoch": 1.7641518919462413, + "grad_norm": 0.3493230640888214, + "learning_rate": 1.7714477211796249e-06, + "loss": 0.523811674118042, + "step": 6580 + }, + { + "epoch": 1.769514361363408, + "grad_norm": 0.1742691546678543, + "learning_rate": 1.7312332439678286e-06, + "loss": 0.5276295661926269, + "step": 6600 + }, + { + "epoch": 1.7748768307805745, + "grad_norm": 0.16134823858737946, + "learning_rate": 1.6910187667560323e-06, + "loss": 0.5352637290954589, + "step": 6620 + }, + { + "epoch": 1.780239300197741, + "grad_norm": 0.20977018773555756, + "learning_rate": 1.650804289544236e-06, + "loss": 0.4955774784088135, + "step": 6640 + }, + { + "epoch": 1.7856017696149076, + "grad_norm": 0.20511005818843842, + "learning_rate": 1.6105898123324397e-06, + "loss": 0.48643174171447756, + "step": 6660 + }, + { + "epoch": 1.7909642390320744, + "grad_norm": 0.23870044946670532, + "learning_rate": 1.5703753351206434e-06, + "loss": 0.4673162460327148, + "step": 6680 + }, + { + "epoch": 1.796326708449241, + "grad_norm": 0.21660065650939941, + "learning_rate": 1.5301608579088473e-06, + "loss": 0.5381903648376465, + "step": 6700 + }, + { + "epoch": 1.8016891778664075, + "grad_norm": 0.26977139711380005, + "learning_rate": 1.489946380697051e-06, + "loss": 0.42094998359680175, + "step": 6720 + }, + { + "epoch": 1.807051647283574, + "grad_norm": 0.2088550478219986, + "learning_rate": 1.4497319034852549e-06, + "loss": 0.49211792945861815, + "step": 6740 + }, + { + "epoch": 1.8124141167007406, + "grad_norm": 0.18141885101795197, + "learning_rate": 1.4095174262734586e-06, + "loss": 0.46572179794311525, + "step": 6760 + }, + { + "epoch": 1.8177765861179074, + "grad_norm": 0.2200685739517212, + "learning_rate": 1.3693029490616623e-06, + "loss": 0.4996177196502686, + "step": 6780 + }, + { + "epoch": 1.823139055535074, + "grad_norm": 0.19545452296733856, + "learning_rate": 1.329088471849866e-06, + "loss": 0.4731945514678955, + "step": 6800 + }, + { + "epoch": 1.8285015249522405, + "grad_norm": 0.2239731252193451, + "learning_rate": 1.2888739946380697e-06, + "loss": 0.47544050216674805, + "step": 6820 + }, + { + "epoch": 1.833863994369407, + "grad_norm": 0.22336581349372864, + "learning_rate": 1.2486595174262734e-06, + "loss": 0.47878737449645997, + "step": 6840 + }, + { + "epoch": 1.8392264637865736, + "grad_norm": 0.20921571552753448, + "learning_rate": 1.2084450402144773e-06, + "loss": 0.41347403526306153, + "step": 6860 + }, + { + "epoch": 1.8445889332037404, + "grad_norm": 0.1577194333076477, + "learning_rate": 1.168230563002681e-06, + "loss": 0.5464958667755127, + "step": 6880 + }, + { + "epoch": 1.849951402620907, + "grad_norm": 0.1477355808019638, + "learning_rate": 1.1280160857908849e-06, + "loss": 0.48548617362976076, + "step": 6900 + }, + { + "epoch": 1.8553138720380735, + "grad_norm": 0.22352682054042816, + "learning_rate": 1.0878016085790886e-06, + "loss": 0.4518588542938232, + "step": 6920 + }, + { + "epoch": 1.8606763414552403, + "grad_norm": 0.19822706282138824, + "learning_rate": 1.0475871313672923e-06, + "loss": 0.4190972805023193, + "step": 6940 + }, + { + "epoch": 1.8660388108724066, + "grad_norm": 0.20670010149478912, + "learning_rate": 1.007372654155496e-06, + "loss": 0.5045090675354004, + "step": 6960 + }, + { + "epoch": 1.8714012802895734, + "grad_norm": 0.2154514342546463, + "learning_rate": 9.671581769436997e-07, + "loss": 0.4472982883453369, + "step": 6980 + }, + { + "epoch": 1.87676374970674, + "grad_norm": 0.19451302289962769, + "learning_rate": 9.269436997319035e-07, + "loss": 0.4328409194946289, + "step": 7000 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.620995010015519e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-7000/training_args.bin b/checkpoint-7000/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-7000/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-7200/README.md b/checkpoint-7200/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-7200/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-7200/adapter_config.json b/checkpoint-7200/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-7200/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-7200/adapter_model.safetensors b/checkpoint-7200/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9f055fc8abdaec3c7b782514e3c1542994d68cff --- /dev/null +++ b/checkpoint-7200/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5008c50618f5d0ed327d88a93d41f3cdc02a97240319db261643e05167eb465a +size 50360752 diff --git a/checkpoint-7200/chat_template.jinja b/checkpoint-7200/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-7200/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-7200/optimizer.pt b/checkpoint-7200/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..38adf2b7caee9e427f28df019f90fe63b53e8a87 --- /dev/null +++ b/checkpoint-7200/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a8585e77057ed91d1a6749c6045fe9717378b70a3650a76c0fecbc92a19f3735 +size 100828235 diff --git a/checkpoint-7200/rng_state.pth b/checkpoint-7200/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..6dbe3dd1a22eb0152d28f4f426a96fb325e714c2 --- /dev/null +++ b/checkpoint-7200/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:993ebcc8811681c7cc30a6b9b00d1bad83cd6154bfbb94372a71b16f140254bf +size 14645 diff --git a/checkpoint-7200/scheduler.pt b/checkpoint-7200/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..4a2886cad7602acc88568e1df58d4ec35ec2fb85 --- /dev/null +++ b/checkpoint-7200/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bdfe5d3784208321e032d0e5bb471a989eb069b864e2db02885ffb5be3116a81 +size 1465 diff --git a/checkpoint-7200/tokenizer.json b/checkpoint-7200/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-7200/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-7200/tokenizer_config.json b/checkpoint-7200/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-7200/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-7200/trainer_state.json b/checkpoint-7200/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..869ea6a497cdbcaa6d64c2e331f311b08a746887 --- /dev/null +++ b/checkpoint-7200/trainer_state.json @@ -0,0 +1,2554 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.930388443878406, + "eval_steps": 500, + "global_step": 7200, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + }, + { + "epoch": 1.6676274424372424, + "grad_norm": 0.22097784280776978, + "learning_rate": 2.4953083109919572e-06, + "loss": 0.43366107940673826, + "step": 6220 + }, + { + "epoch": 1.672989911854409, + "grad_norm": 0.2299281805753708, + "learning_rate": 2.455093833780161e-06, + "loss": 0.5145821094512939, + "step": 6240 + }, + { + "epoch": 1.6783523812715755, + "grad_norm": 0.2384844720363617, + "learning_rate": 2.4148793565683646e-06, + "loss": 0.459308385848999, + "step": 6260 + }, + { + "epoch": 1.6837148506887423, + "grad_norm": 0.24471035599708557, + "learning_rate": 2.3746648793565683e-06, + "loss": 0.4676504611968994, + "step": 6280 + }, + { + "epoch": 1.6890773201059086, + "grad_norm": 0.24419866502285004, + "learning_rate": 2.334450402144772e-06, + "loss": 0.4745138168334961, + "step": 6300 + }, + { + "epoch": 1.6944397895230754, + "grad_norm": 0.15896575152873993, + "learning_rate": 2.294235924932976e-06, + "loss": 0.5073649883270264, + "step": 6320 + }, + { + "epoch": 1.699802258940242, + "grad_norm": 0.26504868268966675, + "learning_rate": 2.25402144772118e-06, + "loss": 0.4534353733062744, + "step": 6340 + }, + { + "epoch": 1.7051647283574085, + "grad_norm": 0.2461850792169571, + "learning_rate": 2.2138069705093836e-06, + "loss": 0.4862947940826416, + "step": 6360 + }, + { + "epoch": 1.7105271977745753, + "grad_norm": 0.17332817614078522, + "learning_rate": 2.1735924932975873e-06, + "loss": 0.5049370765686035, + "step": 6380 + }, + { + "epoch": 1.7158896671917419, + "grad_norm": 0.19762548804283142, + "learning_rate": 2.133378016085791e-06, + "loss": 0.5272616386413574, + "step": 6400 + }, + { + "epoch": 1.7212521366089084, + "grad_norm": 0.23265399038791656, + "learning_rate": 2.0931635388739946e-06, + "loss": 0.47600841522216797, + "step": 6420 + }, + { + "epoch": 1.726614606026075, + "grad_norm": 0.20868578553199768, + "learning_rate": 2.0529490616621983e-06, + "loss": 0.5027226448059082, + "step": 6440 + }, + { + "epoch": 1.7319770754432415, + "grad_norm": 0.2851981520652771, + "learning_rate": 2.012734584450402e-06, + "loss": 0.5288124561309815, + "step": 6460 + }, + { + "epoch": 1.7373395448604083, + "grad_norm": 0.20086587965488434, + "learning_rate": 1.9725201072386057e-06, + "loss": 0.4625516891479492, + "step": 6480 + }, + { + "epoch": 1.7427020142775749, + "grad_norm": 0.24060192704200745, + "learning_rate": 1.93230563002681e-06, + "loss": 0.4843903541564941, + "step": 6500 + }, + { + "epoch": 1.7480644836947414, + "grad_norm": 0.33561915159225464, + "learning_rate": 1.8920911528150133e-06, + "loss": 0.4823720932006836, + "step": 6520 + }, + { + "epoch": 1.753426953111908, + "grad_norm": 0.2510465383529663, + "learning_rate": 1.851876675603217e-06, + "loss": 0.46517143249511717, + "step": 6540 + }, + { + "epoch": 1.7587894225290746, + "grad_norm": 0.2631177604198456, + "learning_rate": 1.811662198391421e-06, + "loss": 0.5004732131958007, + "step": 6560 + }, + { + "epoch": 1.7641518919462413, + "grad_norm": 0.3493230640888214, + "learning_rate": 1.7714477211796249e-06, + "loss": 0.523811674118042, + "step": 6580 + }, + { + "epoch": 1.769514361363408, + "grad_norm": 0.1742691546678543, + "learning_rate": 1.7312332439678286e-06, + "loss": 0.5276295661926269, + "step": 6600 + }, + { + "epoch": 1.7748768307805745, + "grad_norm": 0.16134823858737946, + "learning_rate": 1.6910187667560323e-06, + "loss": 0.5352637290954589, + "step": 6620 + }, + { + "epoch": 1.780239300197741, + "grad_norm": 0.20977018773555756, + "learning_rate": 1.650804289544236e-06, + "loss": 0.4955774784088135, + "step": 6640 + }, + { + "epoch": 1.7856017696149076, + "grad_norm": 0.20511005818843842, + "learning_rate": 1.6105898123324397e-06, + "loss": 0.48643174171447756, + "step": 6660 + }, + { + "epoch": 1.7909642390320744, + "grad_norm": 0.23870044946670532, + "learning_rate": 1.5703753351206434e-06, + "loss": 0.4673162460327148, + "step": 6680 + }, + { + "epoch": 1.796326708449241, + "grad_norm": 0.21660065650939941, + "learning_rate": 1.5301608579088473e-06, + "loss": 0.5381903648376465, + "step": 6700 + }, + { + "epoch": 1.8016891778664075, + "grad_norm": 0.26977139711380005, + "learning_rate": 1.489946380697051e-06, + "loss": 0.42094998359680175, + "step": 6720 + }, + { + "epoch": 1.807051647283574, + "grad_norm": 0.2088550478219986, + "learning_rate": 1.4497319034852549e-06, + "loss": 0.49211792945861815, + "step": 6740 + }, + { + "epoch": 1.8124141167007406, + "grad_norm": 0.18141885101795197, + "learning_rate": 1.4095174262734586e-06, + "loss": 0.46572179794311525, + "step": 6760 + }, + { + "epoch": 1.8177765861179074, + "grad_norm": 0.2200685739517212, + "learning_rate": 1.3693029490616623e-06, + "loss": 0.4996177196502686, + "step": 6780 + }, + { + "epoch": 1.823139055535074, + "grad_norm": 0.19545452296733856, + "learning_rate": 1.329088471849866e-06, + "loss": 0.4731945514678955, + "step": 6800 + }, + { + "epoch": 1.8285015249522405, + "grad_norm": 0.2239731252193451, + "learning_rate": 1.2888739946380697e-06, + "loss": 0.47544050216674805, + "step": 6820 + }, + { + "epoch": 1.833863994369407, + "grad_norm": 0.22336581349372864, + "learning_rate": 1.2486595174262734e-06, + "loss": 0.47878737449645997, + "step": 6840 + }, + { + "epoch": 1.8392264637865736, + "grad_norm": 0.20921571552753448, + "learning_rate": 1.2084450402144773e-06, + "loss": 0.41347403526306153, + "step": 6860 + }, + { + "epoch": 1.8445889332037404, + "grad_norm": 0.1577194333076477, + "learning_rate": 1.168230563002681e-06, + "loss": 0.5464958667755127, + "step": 6880 + }, + { + "epoch": 1.849951402620907, + "grad_norm": 0.1477355808019638, + "learning_rate": 1.1280160857908849e-06, + "loss": 0.48548617362976076, + "step": 6900 + }, + { + "epoch": 1.8553138720380735, + "grad_norm": 0.22352682054042816, + "learning_rate": 1.0878016085790886e-06, + "loss": 0.4518588542938232, + "step": 6920 + }, + { + "epoch": 1.8606763414552403, + "grad_norm": 0.19822706282138824, + "learning_rate": 1.0475871313672923e-06, + "loss": 0.4190972805023193, + "step": 6940 + }, + { + "epoch": 1.8660388108724066, + "grad_norm": 0.20670010149478912, + "learning_rate": 1.007372654155496e-06, + "loss": 0.5045090675354004, + "step": 6960 + }, + { + "epoch": 1.8714012802895734, + "grad_norm": 0.2154514342546463, + "learning_rate": 9.671581769436997e-07, + "loss": 0.4472982883453369, + "step": 6980 + }, + { + "epoch": 1.87676374970674, + "grad_norm": 0.19451302289962769, + "learning_rate": 9.269436997319035e-07, + "loss": 0.4328409194946289, + "step": 7000 + }, + { + "epoch": 1.8821262191239065, + "grad_norm": 0.20980985462665558, + "learning_rate": 8.867292225201073e-07, + "loss": 0.43312845230102537, + "step": 7020 + }, + { + "epoch": 1.8874886885410733, + "grad_norm": 0.19927652180194855, + "learning_rate": 8.46514745308311e-07, + "loss": 0.5255829811096191, + "step": 7040 + }, + { + "epoch": 1.8928511579582397, + "grad_norm": 0.2869090437889099, + "learning_rate": 8.063002680965148e-07, + "loss": 0.5301108360290527, + "step": 7060 + }, + { + "epoch": 1.8982136273754064, + "grad_norm": 0.16662724316120148, + "learning_rate": 7.660857908847185e-07, + "loss": 0.48042588233947753, + "step": 7080 + }, + { + "epoch": 1.903576096792573, + "grad_norm": 0.21045279502868652, + "learning_rate": 7.258713136729223e-07, + "loss": 0.4943391799926758, + "step": 7100 + }, + { + "epoch": 1.9089385662097396, + "grad_norm": 0.18638195097446442, + "learning_rate": 6.85656836461126e-07, + "loss": 0.48406500816345216, + "step": 7120 + }, + { + "epoch": 1.9143010356269063, + "grad_norm": 0.15428832173347473, + "learning_rate": 6.454423592493298e-07, + "loss": 0.5084923267364502, + "step": 7140 + }, + { + "epoch": 1.9196635050440727, + "grad_norm": 0.2415294051170349, + "learning_rate": 6.052278820375336e-07, + "loss": 0.5026498317718506, + "step": 7160 + }, + { + "epoch": 1.9250259744612395, + "grad_norm": 0.23021087050437927, + "learning_rate": 5.650134048257373e-07, + "loss": 0.5294596195220947, + "step": 7180 + }, + { + "epoch": 1.930388443878406, + "grad_norm": 0.21689893305301666, + "learning_rate": 5.24798927613941e-07, + "loss": 0.4569683074951172, + "step": 7200 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.868448234493706e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-7200/training_args.bin b/checkpoint-7200/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-7200/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-7400/README.md b/checkpoint-7400/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-7400/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-7400/adapter_config.json b/checkpoint-7400/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-7400/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-7400/adapter_model.safetensors b/checkpoint-7400/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2ac0c41fcf74f204ac49ca5afd4cf72f8acdca3b --- /dev/null +++ b/checkpoint-7400/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:abc704f30a112227b3ce512e37c72893b188051a382e1329a88f043367b5ae7c +size 50360752 diff --git a/checkpoint-7400/chat_template.jinja b/checkpoint-7400/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-7400/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-7400/optimizer.pt b/checkpoint-7400/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..73ed6723d31a896540c4077a7a1c390649d6ad1b --- /dev/null +++ b/checkpoint-7400/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a32478a2d44562e34964ded226b738d22fefd30abb28b52a1e2912131a197520 +size 100828235 diff --git a/checkpoint-7400/rng_state.pth b/checkpoint-7400/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a3535d199a9425419fc2a0776faa98eea7cf24e0 --- /dev/null +++ b/checkpoint-7400/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:80bb9c70314145190b1bed112a68c6d20e2cf4d8364307a86eaaec3f257f16ad +size 14645 diff --git a/checkpoint-7400/scheduler.pt b/checkpoint-7400/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..c3048c9532c6d2653b4ea9c2c088faf5054997c1 --- /dev/null +++ b/checkpoint-7400/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:00c1d3473202ef174820e054d21dcd57620585a9ad949f4a2124d723b5c9572f +size 1465 diff --git a/checkpoint-7400/tokenizer.json b/checkpoint-7400/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-7400/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-7400/tokenizer_config.json b/checkpoint-7400/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-7400/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-7400/trainer_state.json b/checkpoint-7400/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..dbdf241193f77d5519a0fd8883f0f1b62f13646a --- /dev/null +++ b/checkpoint-7400/trainer_state.json @@ -0,0 +1,2624 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.984013138050072, + "eval_steps": 500, + "global_step": 7400, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + }, + { + "epoch": 1.6676274424372424, + "grad_norm": 0.22097784280776978, + "learning_rate": 2.4953083109919572e-06, + "loss": 0.43366107940673826, + "step": 6220 + }, + { + "epoch": 1.672989911854409, + "grad_norm": 0.2299281805753708, + "learning_rate": 2.455093833780161e-06, + "loss": 0.5145821094512939, + "step": 6240 + }, + { + "epoch": 1.6783523812715755, + "grad_norm": 0.2384844720363617, + "learning_rate": 2.4148793565683646e-06, + "loss": 0.459308385848999, + "step": 6260 + }, + { + "epoch": 1.6837148506887423, + "grad_norm": 0.24471035599708557, + "learning_rate": 2.3746648793565683e-06, + "loss": 0.4676504611968994, + "step": 6280 + }, + { + "epoch": 1.6890773201059086, + "grad_norm": 0.24419866502285004, + "learning_rate": 2.334450402144772e-06, + "loss": 0.4745138168334961, + "step": 6300 + }, + { + "epoch": 1.6944397895230754, + "grad_norm": 0.15896575152873993, + "learning_rate": 2.294235924932976e-06, + "loss": 0.5073649883270264, + "step": 6320 + }, + { + "epoch": 1.699802258940242, + "grad_norm": 0.26504868268966675, + "learning_rate": 2.25402144772118e-06, + "loss": 0.4534353733062744, + "step": 6340 + }, + { + "epoch": 1.7051647283574085, + "grad_norm": 0.2461850792169571, + "learning_rate": 2.2138069705093836e-06, + "loss": 0.4862947940826416, + "step": 6360 + }, + { + "epoch": 1.7105271977745753, + "grad_norm": 0.17332817614078522, + "learning_rate": 2.1735924932975873e-06, + "loss": 0.5049370765686035, + "step": 6380 + }, + { + "epoch": 1.7158896671917419, + "grad_norm": 0.19762548804283142, + "learning_rate": 2.133378016085791e-06, + "loss": 0.5272616386413574, + "step": 6400 + }, + { + "epoch": 1.7212521366089084, + "grad_norm": 0.23265399038791656, + "learning_rate": 2.0931635388739946e-06, + "loss": 0.47600841522216797, + "step": 6420 + }, + { + "epoch": 1.726614606026075, + "grad_norm": 0.20868578553199768, + "learning_rate": 2.0529490616621983e-06, + "loss": 0.5027226448059082, + "step": 6440 + }, + { + "epoch": 1.7319770754432415, + "grad_norm": 0.2851981520652771, + "learning_rate": 2.012734584450402e-06, + "loss": 0.5288124561309815, + "step": 6460 + }, + { + "epoch": 1.7373395448604083, + "grad_norm": 0.20086587965488434, + "learning_rate": 1.9725201072386057e-06, + "loss": 0.4625516891479492, + "step": 6480 + }, + { + "epoch": 1.7427020142775749, + "grad_norm": 0.24060192704200745, + "learning_rate": 1.93230563002681e-06, + "loss": 0.4843903541564941, + "step": 6500 + }, + { + "epoch": 1.7480644836947414, + "grad_norm": 0.33561915159225464, + "learning_rate": 1.8920911528150133e-06, + "loss": 0.4823720932006836, + "step": 6520 + }, + { + "epoch": 1.753426953111908, + "grad_norm": 0.2510465383529663, + "learning_rate": 1.851876675603217e-06, + "loss": 0.46517143249511717, + "step": 6540 + }, + { + "epoch": 1.7587894225290746, + "grad_norm": 0.2631177604198456, + "learning_rate": 1.811662198391421e-06, + "loss": 0.5004732131958007, + "step": 6560 + }, + { + "epoch": 1.7641518919462413, + "grad_norm": 0.3493230640888214, + "learning_rate": 1.7714477211796249e-06, + "loss": 0.523811674118042, + "step": 6580 + }, + { + "epoch": 1.769514361363408, + "grad_norm": 0.1742691546678543, + "learning_rate": 1.7312332439678286e-06, + "loss": 0.5276295661926269, + "step": 6600 + }, + { + "epoch": 1.7748768307805745, + "grad_norm": 0.16134823858737946, + "learning_rate": 1.6910187667560323e-06, + "loss": 0.5352637290954589, + "step": 6620 + }, + { + "epoch": 1.780239300197741, + "grad_norm": 0.20977018773555756, + "learning_rate": 1.650804289544236e-06, + "loss": 0.4955774784088135, + "step": 6640 + }, + { + "epoch": 1.7856017696149076, + "grad_norm": 0.20511005818843842, + "learning_rate": 1.6105898123324397e-06, + "loss": 0.48643174171447756, + "step": 6660 + }, + { + "epoch": 1.7909642390320744, + "grad_norm": 0.23870044946670532, + "learning_rate": 1.5703753351206434e-06, + "loss": 0.4673162460327148, + "step": 6680 + }, + { + "epoch": 1.796326708449241, + "grad_norm": 0.21660065650939941, + "learning_rate": 1.5301608579088473e-06, + "loss": 0.5381903648376465, + "step": 6700 + }, + { + "epoch": 1.8016891778664075, + "grad_norm": 0.26977139711380005, + "learning_rate": 1.489946380697051e-06, + "loss": 0.42094998359680175, + "step": 6720 + }, + { + "epoch": 1.807051647283574, + "grad_norm": 0.2088550478219986, + "learning_rate": 1.4497319034852549e-06, + "loss": 0.49211792945861815, + "step": 6740 + }, + { + "epoch": 1.8124141167007406, + "grad_norm": 0.18141885101795197, + "learning_rate": 1.4095174262734586e-06, + "loss": 0.46572179794311525, + "step": 6760 + }, + { + "epoch": 1.8177765861179074, + "grad_norm": 0.2200685739517212, + "learning_rate": 1.3693029490616623e-06, + "loss": 0.4996177196502686, + "step": 6780 + }, + { + "epoch": 1.823139055535074, + "grad_norm": 0.19545452296733856, + "learning_rate": 1.329088471849866e-06, + "loss": 0.4731945514678955, + "step": 6800 + }, + { + "epoch": 1.8285015249522405, + "grad_norm": 0.2239731252193451, + "learning_rate": 1.2888739946380697e-06, + "loss": 0.47544050216674805, + "step": 6820 + }, + { + "epoch": 1.833863994369407, + "grad_norm": 0.22336581349372864, + "learning_rate": 1.2486595174262734e-06, + "loss": 0.47878737449645997, + "step": 6840 + }, + { + "epoch": 1.8392264637865736, + "grad_norm": 0.20921571552753448, + "learning_rate": 1.2084450402144773e-06, + "loss": 0.41347403526306153, + "step": 6860 + }, + { + "epoch": 1.8445889332037404, + "grad_norm": 0.1577194333076477, + "learning_rate": 1.168230563002681e-06, + "loss": 0.5464958667755127, + "step": 6880 + }, + { + "epoch": 1.849951402620907, + "grad_norm": 0.1477355808019638, + "learning_rate": 1.1280160857908849e-06, + "loss": 0.48548617362976076, + "step": 6900 + }, + { + "epoch": 1.8553138720380735, + "grad_norm": 0.22352682054042816, + "learning_rate": 1.0878016085790886e-06, + "loss": 0.4518588542938232, + "step": 6920 + }, + { + "epoch": 1.8606763414552403, + "grad_norm": 0.19822706282138824, + "learning_rate": 1.0475871313672923e-06, + "loss": 0.4190972805023193, + "step": 6940 + }, + { + "epoch": 1.8660388108724066, + "grad_norm": 0.20670010149478912, + "learning_rate": 1.007372654155496e-06, + "loss": 0.5045090675354004, + "step": 6960 + }, + { + "epoch": 1.8714012802895734, + "grad_norm": 0.2154514342546463, + "learning_rate": 9.671581769436997e-07, + "loss": 0.4472982883453369, + "step": 6980 + }, + { + "epoch": 1.87676374970674, + "grad_norm": 0.19451302289962769, + "learning_rate": 9.269436997319035e-07, + "loss": 0.4328409194946289, + "step": 7000 + }, + { + "epoch": 1.8821262191239065, + "grad_norm": 0.20980985462665558, + "learning_rate": 8.867292225201073e-07, + "loss": 0.43312845230102537, + "step": 7020 + }, + { + "epoch": 1.8874886885410733, + "grad_norm": 0.19927652180194855, + "learning_rate": 8.46514745308311e-07, + "loss": 0.5255829811096191, + "step": 7040 + }, + { + "epoch": 1.8928511579582397, + "grad_norm": 0.2869090437889099, + "learning_rate": 8.063002680965148e-07, + "loss": 0.5301108360290527, + "step": 7060 + }, + { + "epoch": 1.8982136273754064, + "grad_norm": 0.16662724316120148, + "learning_rate": 7.660857908847185e-07, + "loss": 0.48042588233947753, + "step": 7080 + }, + { + "epoch": 1.903576096792573, + "grad_norm": 0.21045279502868652, + "learning_rate": 7.258713136729223e-07, + "loss": 0.4943391799926758, + "step": 7100 + }, + { + "epoch": 1.9089385662097396, + "grad_norm": 0.18638195097446442, + "learning_rate": 6.85656836461126e-07, + "loss": 0.48406500816345216, + "step": 7120 + }, + { + "epoch": 1.9143010356269063, + "grad_norm": 0.15428832173347473, + "learning_rate": 6.454423592493298e-07, + "loss": 0.5084923267364502, + "step": 7140 + }, + { + "epoch": 1.9196635050440727, + "grad_norm": 0.2415294051170349, + "learning_rate": 6.052278820375336e-07, + "loss": 0.5026498317718506, + "step": 7160 + }, + { + "epoch": 1.9250259744612395, + "grad_norm": 0.23021087050437927, + "learning_rate": 5.650134048257373e-07, + "loss": 0.5294596195220947, + "step": 7180 + }, + { + "epoch": 1.930388443878406, + "grad_norm": 0.21689893305301666, + "learning_rate": 5.24798927613941e-07, + "loss": 0.4569683074951172, + "step": 7200 + }, + { + "epoch": 1.9357509132955726, + "grad_norm": 0.25458022952079773, + "learning_rate": 4.845844504021448e-07, + "loss": 0.4905412197113037, + "step": 7220 + }, + { + "epoch": 1.9411133827127394, + "grad_norm": 0.20990079641342163, + "learning_rate": 4.4436997319034854e-07, + "loss": 0.49599390029907225, + "step": 7240 + }, + { + "epoch": 1.9464758521299057, + "grad_norm": 0.2381497025489807, + "learning_rate": 4.041554959785523e-07, + "loss": 0.5149903774261475, + "step": 7260 + }, + { + "epoch": 1.9518383215470725, + "grad_norm": 0.29006749391555786, + "learning_rate": 3.6394101876675604e-07, + "loss": 0.5063531875610352, + "step": 7280 + }, + { + "epoch": 1.957200790964239, + "grad_norm": 0.20159471035003662, + "learning_rate": 3.237265415549598e-07, + "loss": 0.5062472343444824, + "step": 7300 + }, + { + "epoch": 1.9625632603814056, + "grad_norm": 0.21182367205619812, + "learning_rate": 2.8351206434316354e-07, + "loss": 0.49964118003845215, + "step": 7320 + }, + { + "epoch": 1.9679257297985724, + "grad_norm": 0.28606927394866943, + "learning_rate": 2.432975871313673e-07, + "loss": 0.48647170066833495, + "step": 7340 + }, + { + "epoch": 1.9732881992157387, + "grad_norm": 0.2779375910758972, + "learning_rate": 2.0308310991957104e-07, + "loss": 0.4816920280456543, + "step": 7360 + }, + { + "epoch": 1.9786506686329055, + "grad_norm": 0.20412451028823853, + "learning_rate": 1.628686327077748e-07, + "loss": 0.5177248477935791, + "step": 7380 + }, + { + "epoch": 1.984013138050072, + "grad_norm": 0.19861914217472076, + "learning_rate": 1.2265415549597854e-07, + "loss": 0.5265318870544433, + "step": 7400 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.118750722760274e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-7400/training_args.bin b/checkpoint-7400/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-7400/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-7460/README.md b/checkpoint-7460/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-7460/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-7460/adapter_config.json b/checkpoint-7460/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-7460/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-7460/adapter_model.safetensors b/checkpoint-7460/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2d56dd52e04576c533bceb18ec394b50943e9d91 --- /dev/null +++ b/checkpoint-7460/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e45b98549709b9485f32de70adf0f96edce8ca97b599382c87a97973290a9c87 +size 50360752 diff --git a/checkpoint-7460/chat_template.jinja b/checkpoint-7460/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-7460/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-7460/optimizer.pt b/checkpoint-7460/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..1a2eeab20f259c7ed54f824a514fd2254a47fbc1 --- /dev/null +++ b/checkpoint-7460/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4e64fa3c707df178cda23d639abb1f1cafc21e2dbf509d522cfd8f580758a69 +size 100828235 diff --git a/checkpoint-7460/rng_state.pth b/checkpoint-7460/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..38d00ee5d06e3f13ef1759ccbdb3543090dd7260 --- /dev/null +++ b/checkpoint-7460/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:823c0df99ec859ef099d61782766170fdc56bce9968f4ae05385de24a019d319 +size 14645 diff --git a/checkpoint-7460/scheduler.pt b/checkpoint-7460/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d30cf94abd4985514130e141607ae9ab5086ea1e --- /dev/null +++ b/checkpoint-7460/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0e08f38c509bf4e9ef67661f9a81b33e13d2563c8cb40e2dfda7539d931c734d +size 1465 diff --git a/checkpoint-7460/tokenizer.json b/checkpoint-7460/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-7460/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-7460/tokenizer_config.json b/checkpoint-7460/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-7460/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-7460/trainer_state.json b/checkpoint-7460/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..078db794fc37dbb4d6ec0624d30d01b51bab32e5 --- /dev/null +++ b/checkpoint-7460/trainer_state.json @@ -0,0 +1,2645 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.0, + "eval_steps": 500, + "global_step": 7460, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + }, + { + "epoch": 0.21986124610383082, + "grad_norm": 0.08678701519966125, + "learning_rate": 1.3353217158176944e-05, + "loss": 0.5999309062957764, + "step": 820 + }, + { + "epoch": 0.22522371552099743, + "grad_norm": 0.12222636491060257, + "learning_rate": 1.3313002680965148e-05, + "loss": 0.5421838760375977, + "step": 840 + }, + { + "epoch": 0.23058618493816402, + "grad_norm": 0.11634483933448792, + "learning_rate": 1.3272788203753352e-05, + "loss": 0.6069926261901856, + "step": 860 + }, + { + "epoch": 0.23594865435533063, + "grad_norm": 0.12163955718278885, + "learning_rate": 1.3232573726541556e-05, + "loss": 0.5558357238769531, + "step": 880 + }, + { + "epoch": 0.24131112377249722, + "grad_norm": 0.13140572607517242, + "learning_rate": 1.319235924932976e-05, + "loss": 0.5537341117858887, + "step": 900 + }, + { + "epoch": 0.24667359318966384, + "grad_norm": 0.1295424848794937, + "learning_rate": 1.3152144772117963e-05, + "loss": 0.5734247684478759, + "step": 920 + }, + { + "epoch": 0.2520360626068304, + "grad_norm": 0.08855397999286652, + "learning_rate": 1.3111930294906167e-05, + "loss": 0.5499854564666748, + "step": 940 + }, + { + "epoch": 0.25739853202399704, + "grad_norm": 0.10895389318466187, + "learning_rate": 1.307171581769437e-05, + "loss": 0.4994966506958008, + "step": 960 + }, + { + "epoch": 0.26276100144116366, + "grad_norm": 0.10110122710466385, + "learning_rate": 1.3031501340482574e-05, + "loss": 0.5803254604339599, + "step": 980 + }, + { + "epoch": 0.26812347085833027, + "grad_norm": 0.1323656141757965, + "learning_rate": 1.2991286863270778e-05, + "loss": 0.5268758773803711, + "step": 1000 + }, + { + "epoch": 0.2734859402754969, + "grad_norm": 0.09068968147039413, + "learning_rate": 1.2951072386058981e-05, + "loss": 0.5150487899780274, + "step": 1020 + }, + { + "epoch": 0.27884840969266345, + "grad_norm": 0.11400057375431061, + "learning_rate": 1.2910857908847185e-05, + "loss": 0.5365507125854492, + "step": 1040 + }, + { + "epoch": 0.28421087910983006, + "grad_norm": 0.14133770763874054, + "learning_rate": 1.2870643431635389e-05, + "loss": 0.5134270668029786, + "step": 1060 + }, + { + "epoch": 0.2895733485269967, + "grad_norm": 0.14621631801128387, + "learning_rate": 1.2830428954423593e-05, + "loss": 0.5870331287384033, + "step": 1080 + }, + { + "epoch": 0.2949358179441633, + "grad_norm": 0.09397239238023758, + "learning_rate": 1.2790214477211796e-05, + "loss": 0.5265964984893798, + "step": 1100 + }, + { + "epoch": 0.3002982873613299, + "grad_norm": 0.13457220792770386, + "learning_rate": 1.275e-05, + "loss": 0.541674280166626, + "step": 1120 + }, + { + "epoch": 0.3056607567784965, + "grad_norm": 0.11553078144788742, + "learning_rate": 1.2709785522788204e-05, + "loss": 0.5721035003662109, + "step": 1140 + }, + { + "epoch": 0.3110232261956631, + "grad_norm": 0.08464279770851135, + "learning_rate": 1.2669571045576407e-05, + "loss": 0.5242496967315674, + "step": 1160 + }, + { + "epoch": 0.3163856956128297, + "grad_norm": 0.11578533798456192, + "learning_rate": 1.2629356568364611e-05, + "loss": 0.5268265724182128, + "step": 1180 + }, + { + "epoch": 0.3217481650299963, + "grad_norm": 0.10422660410404205, + "learning_rate": 1.2589142091152815e-05, + "loss": 0.5755553722381592, + "step": 1200 + }, + { + "epoch": 0.32711063444716293, + "grad_norm": 0.1601565182209015, + "learning_rate": 1.2548927613941018e-05, + "loss": 0.572784423828125, + "step": 1220 + }, + { + "epoch": 0.33247310386432954, + "grad_norm": 0.1435895711183548, + "learning_rate": 1.2508713136729222e-05, + "loss": 0.4759331703186035, + "step": 1240 + }, + { + "epoch": 0.3378355732814961, + "grad_norm": 0.13164320588111877, + "learning_rate": 1.2468498659517426e-05, + "loss": 0.5674447059631348, + "step": 1260 + }, + { + "epoch": 0.3431980426986627, + "grad_norm": 0.17907585203647614, + "learning_rate": 1.242828418230563e-05, + "loss": 0.5384601593017578, + "step": 1280 + }, + { + "epoch": 0.34856051211582934, + "grad_norm": 0.1515372097492218, + "learning_rate": 1.2388069705093833e-05, + "loss": 0.5154921531677246, + "step": 1300 + }, + { + "epoch": 0.35392298153299595, + "grad_norm": 0.13605119287967682, + "learning_rate": 1.2347855227882037e-05, + "loss": 0.5586633205413818, + "step": 1320 + }, + { + "epoch": 0.35928545095016257, + "grad_norm": 0.12003476917743683, + "learning_rate": 1.230764075067024e-05, + "loss": 0.5512509822845459, + "step": 1340 + }, + { + "epoch": 0.3646479203673292, + "grad_norm": 0.11852169036865234, + "learning_rate": 1.2267426273458444e-05, + "loss": 0.5680348873138428, + "step": 1360 + }, + { + "epoch": 0.37001038978449574, + "grad_norm": 0.16344694793224335, + "learning_rate": 1.2227211796246648e-05, + "loss": 0.5669443130493164, + "step": 1380 + }, + { + "epoch": 0.37537285920166236, + "grad_norm": 0.11730384081602097, + "learning_rate": 1.2186997319034852e-05, + "loss": 0.5089732646942139, + "step": 1400 + }, + { + "epoch": 0.38073532861882897, + "grad_norm": 0.1063583567738533, + "learning_rate": 1.2146782841823055e-05, + "loss": 0.5337563037872315, + "step": 1420 + }, + { + "epoch": 0.3860977980359956, + "grad_norm": 0.12790119647979736, + "learning_rate": 1.2106568364611259e-05, + "loss": 0.5077777862548828, + "step": 1440 + }, + { + "epoch": 0.3914602674531622, + "grad_norm": 0.1386743038892746, + "learning_rate": 1.2066353887399463e-05, + "loss": 0.5521824836730957, + "step": 1460 + }, + { + "epoch": 0.39682273687032876, + "grad_norm": 0.0992259532213211, + "learning_rate": 1.2026139410187666e-05, + "loss": 0.554673147201538, + "step": 1480 + }, + { + "epoch": 0.4021852062874954, + "grad_norm": 0.15981841087341309, + "learning_rate": 1.1985924932975872e-05, + "loss": 0.5779122352600098, + "step": 1500 + }, + { + "epoch": 0.407547675704662, + "grad_norm": 0.19671906530857086, + "learning_rate": 1.1945710455764076e-05, + "loss": 0.5743378162384033, + "step": 1520 + }, + { + "epoch": 0.4129101451218286, + "grad_norm": 0.10725795477628708, + "learning_rate": 1.190549597855228e-05, + "loss": 0.523157787322998, + "step": 1540 + }, + { + "epoch": 0.4182726145389952, + "grad_norm": 0.14457851648330688, + "learning_rate": 1.1865281501340483e-05, + "loss": 0.5441864490509033, + "step": 1560 + }, + { + "epoch": 0.42363508395616184, + "grad_norm": 0.15479697287082672, + "learning_rate": 1.1825067024128687e-05, + "loss": 0.6409400463104248, + "step": 1580 + }, + { + "epoch": 0.4289975533733284, + "grad_norm": 0.11132492870092392, + "learning_rate": 1.178485254691689e-05, + "loss": 0.5462933540344238, + "step": 1600 + }, + { + "epoch": 0.434360022790495, + "grad_norm": 0.11062806099653244, + "learning_rate": 1.1744638069705094e-05, + "loss": 0.5428354740142822, + "step": 1620 + }, + { + "epoch": 0.43972249220766163, + "grad_norm": 0.1327652931213379, + "learning_rate": 1.1704423592493298e-05, + "loss": 0.5324414253234864, + "step": 1640 + }, + { + "epoch": 0.44508496162482825, + "grad_norm": 0.1209583580493927, + "learning_rate": 1.1664209115281501e-05, + "loss": 0.5270706176757812, + "step": 1660 + }, + { + "epoch": 0.45044743104199486, + "grad_norm": 0.11154980212450027, + "learning_rate": 1.1623994638069705e-05, + "loss": 0.525149154663086, + "step": 1680 + }, + { + "epoch": 0.4558099004591614, + "grad_norm": 0.14099697768688202, + "learning_rate": 1.158378016085791e-05, + "loss": 0.5981990814208984, + "step": 1700 + }, + { + "epoch": 0.46117236987632804, + "grad_norm": 0.11787982285022736, + "learning_rate": 1.1543565683646114e-05, + "loss": 0.5327546119689941, + "step": 1720 + }, + { + "epoch": 0.46653483929349465, + "grad_norm": 0.12584130465984344, + "learning_rate": 1.1503351206434318e-05, + "loss": 0.5126790046691895, + "step": 1740 + }, + { + "epoch": 0.47189730871066127, + "grad_norm": 0.16248232126235962, + "learning_rate": 1.1463136729222522e-05, + "loss": 0.5697287082672119, + "step": 1760 + }, + { + "epoch": 0.4772597781278279, + "grad_norm": 0.14940819144248962, + "learning_rate": 1.1422922252010725e-05, + "loss": 0.5015492916107178, + "step": 1780 + }, + { + "epoch": 0.48262224754499444, + "grad_norm": 0.1647220402956009, + "learning_rate": 1.1382707774798929e-05, + "loss": 0.5097331523895263, + "step": 1800 + }, + { + "epoch": 0.48798471696216106, + "grad_norm": 0.12255030870437622, + "learning_rate": 1.1342493297587133e-05, + "loss": 0.5670981407165527, + "step": 1820 + }, + { + "epoch": 0.4933471863793277, + "grad_norm": 0.1160770058631897, + "learning_rate": 1.1302278820375336e-05, + "loss": 0.5236512660980225, + "step": 1840 + }, + { + "epoch": 0.4987096557964943, + "grad_norm": 0.21711941063404083, + "learning_rate": 1.126206434316354e-05, + "loss": 0.5926671504974366, + "step": 1860 + }, + { + "epoch": 0.5040721252136608, + "grad_norm": 0.16682052612304688, + "learning_rate": 1.1221849865951744e-05, + "loss": 0.5240281581878662, + "step": 1880 + }, + { + "epoch": 0.5094345946308275, + "grad_norm": 0.16348475217819214, + "learning_rate": 1.1181635388739948e-05, + "loss": 0.5574026107788086, + "step": 1900 + }, + { + "epoch": 0.5147970640479941, + "grad_norm": 0.17506958544254303, + "learning_rate": 1.1141420911528151e-05, + "loss": 0.5592098236083984, + "step": 1920 + }, + { + "epoch": 0.5201595334651608, + "grad_norm": 0.1784403771162033, + "learning_rate": 1.1101206434316355e-05, + "loss": 0.5189618110656739, + "step": 1940 + }, + { + "epoch": 0.5255220028823273, + "grad_norm": 0.17252163589000702, + "learning_rate": 1.1060991957104559e-05, + "loss": 0.5126346111297607, + "step": 1960 + }, + { + "epoch": 0.5308844722994939, + "grad_norm": 0.12690365314483643, + "learning_rate": 1.1020777479892762e-05, + "loss": 0.5473652362823487, + "step": 1980 + }, + { + "epoch": 0.5362469417166605, + "grad_norm": 0.1284744292497635, + "learning_rate": 1.0980563002680966e-05, + "loss": 0.5309309482574462, + "step": 2000 + }, + { + "epoch": 0.5416094111338271, + "grad_norm": 0.1850503385066986, + "learning_rate": 1.094034852546917e-05, + "loss": 0.5636833190917969, + "step": 2020 + }, + { + "epoch": 0.5469718805509938, + "grad_norm": 0.1514296680688858, + "learning_rate": 1.0900134048257373e-05, + "loss": 0.5273778915405274, + "step": 2040 + }, + { + "epoch": 0.5523343499681603, + "grad_norm": 0.1502915471792221, + "learning_rate": 1.0859919571045577e-05, + "loss": 0.6000364780426025, + "step": 2060 + }, + { + "epoch": 0.5576968193853269, + "grad_norm": 0.14147423207759857, + "learning_rate": 1.081970509383378e-05, + "loss": 0.5480428218841553, + "step": 2080 + }, + { + "epoch": 0.5630592888024936, + "grad_norm": 0.13399621844291687, + "learning_rate": 1.0779490616621984e-05, + "loss": 0.513938045501709, + "step": 2100 + }, + { + "epoch": 0.5684217582196601, + "grad_norm": 0.12856991589069366, + "learning_rate": 1.0739276139410188e-05, + "loss": 0.4760735988616943, + "step": 2120 + }, + { + "epoch": 0.5737842276368268, + "grad_norm": 0.15576769411563873, + "learning_rate": 1.0699061662198392e-05, + "loss": 0.5474783420562744, + "step": 2140 + }, + { + "epoch": 0.5791466970539934, + "grad_norm": 0.2024153470993042, + "learning_rate": 1.0658847184986596e-05, + "loss": 0.5309592723846436, + "step": 2160 + }, + { + "epoch": 0.58450916647116, + "grad_norm": 0.13033868372440338, + "learning_rate": 1.06186327077748e-05, + "loss": 0.5345770835876464, + "step": 2180 + }, + { + "epoch": 0.5898716358883266, + "grad_norm": 0.15354423224925995, + "learning_rate": 1.0578418230563003e-05, + "loss": 0.5441046714782715, + "step": 2200 + }, + { + "epoch": 0.5952341053054931, + "grad_norm": 0.19533827900886536, + "learning_rate": 1.0538203753351207e-05, + "loss": 0.547668170928955, + "step": 2220 + }, + { + "epoch": 0.6005965747226598, + "grad_norm": 0.15901635587215424, + "learning_rate": 1.049798927613941e-05, + "loss": 0.5213536739349365, + "step": 2240 + }, + { + "epoch": 0.6059590441398264, + "grad_norm": 0.20392107963562012, + "learning_rate": 1.0457774798927614e-05, + "loss": 0.56328444480896, + "step": 2260 + }, + { + "epoch": 0.611321513556993, + "grad_norm": 0.14985501766204834, + "learning_rate": 1.0417560321715818e-05, + "loss": 0.5592964172363282, + "step": 2280 + }, + { + "epoch": 0.6166839829741596, + "grad_norm": 0.16292506456375122, + "learning_rate": 1.0377345844504021e-05, + "loss": 0.6026081562042236, + "step": 2300 + }, + { + "epoch": 0.6220464523913262, + "grad_norm": 0.2114475965499878, + "learning_rate": 1.0337131367292225e-05, + "loss": 0.5434895992279053, + "step": 2320 + }, + { + "epoch": 0.6274089218084928, + "grad_norm": 0.15036092698574066, + "learning_rate": 1.0296916890080429e-05, + "loss": 0.5241796016693115, + "step": 2340 + }, + { + "epoch": 0.6327713912256594, + "grad_norm": 0.2040790617465973, + "learning_rate": 1.0256702412868633e-05, + "loss": 0.5172519683837891, + "step": 2360 + }, + { + "epoch": 0.6381338606428261, + "grad_norm": 0.15708747506141663, + "learning_rate": 1.0216487935656836e-05, + "loss": 0.49505252838134767, + "step": 2380 + }, + { + "epoch": 0.6434963300599926, + "grad_norm": 0.1831217259168625, + "learning_rate": 1.017627345844504e-05, + "loss": 0.5166856288909912, + "step": 2400 + }, + { + "epoch": 0.6488587994771592, + "grad_norm": 0.23026946187019348, + "learning_rate": 1.0136058981233244e-05, + "loss": 0.5275045394897461, + "step": 2420 + }, + { + "epoch": 0.6542212688943259, + "grad_norm": 0.17848673462867737, + "learning_rate": 1.0095844504021447e-05, + "loss": 0.5764461994171143, + "step": 2440 + }, + { + "epoch": 0.6595837383114924, + "grad_norm": 0.14768671989440918, + "learning_rate": 1.0055630026809651e-05, + "loss": 0.4772446632385254, + "step": 2460 + }, + { + "epoch": 0.6649462077286591, + "grad_norm": 0.11061226576566696, + "learning_rate": 1.0015415549597855e-05, + "loss": 0.4822176456451416, + "step": 2480 + }, + { + "epoch": 0.6703086771458256, + "grad_norm": 0.22382384538650513, + "learning_rate": 9.975201072386058e-06, + "loss": 0.5523125648498535, + "step": 2500 + }, + { + "epoch": 0.6756711465629922, + "grad_norm": 0.1481855809688568, + "learning_rate": 9.934986595174262e-06, + "loss": 0.5522858619689941, + "step": 2520 + }, + { + "epoch": 0.6810336159801589, + "grad_norm": 0.16584496200084686, + "learning_rate": 9.894772117962466e-06, + "loss": 0.5220115661621094, + "step": 2540 + }, + { + "epoch": 0.6863960853973254, + "grad_norm": 0.24747292697429657, + "learning_rate": 9.85455764075067e-06, + "loss": 0.5106014728546142, + "step": 2560 + }, + { + "epoch": 0.6917585548144921, + "grad_norm": 0.1886838674545288, + "learning_rate": 9.814343163538873e-06, + "loss": 0.554722261428833, + "step": 2580 + }, + { + "epoch": 0.6971210242316587, + "grad_norm": 0.14403431117534637, + "learning_rate": 9.774128686327077e-06, + "loss": 0.5226208209991455, + "step": 2600 + }, + { + "epoch": 0.7024834936488252, + "grad_norm": 0.1577453911304474, + "learning_rate": 9.73391420911528e-06, + "loss": 0.5295976161956787, + "step": 2620 + }, + { + "epoch": 0.7078459630659919, + "grad_norm": 0.2269749790430069, + "learning_rate": 9.693699731903484e-06, + "loss": 0.5336898803710938, + "step": 2640 + }, + { + "epoch": 0.7132084324831585, + "grad_norm": 0.23890693485736847, + "learning_rate": 9.653485254691688e-06, + "loss": 0.5564133644104003, + "step": 2660 + }, + { + "epoch": 0.7185709019003251, + "grad_norm": 0.19051003456115723, + "learning_rate": 9.613270777479892e-06, + "loss": 0.5483838081359863, + "step": 2680 + }, + { + "epoch": 0.7239333713174917, + "grad_norm": 0.15244685113430023, + "learning_rate": 9.573056300268095e-06, + "loss": 0.5657371520996094, + "step": 2700 + }, + { + "epoch": 0.7292958407346584, + "grad_norm": 0.14131584763526917, + "learning_rate": 9.532841823056299e-06, + "loss": 0.5375633716583252, + "step": 2720 + }, + { + "epoch": 0.7346583101518249, + "grad_norm": 0.15706594288349152, + "learning_rate": 9.492627345844505e-06, + "loss": 0.5774847507476807, + "step": 2740 + }, + { + "epoch": 0.7400207795689915, + "grad_norm": 0.120318703353405, + "learning_rate": 9.452412868632708e-06, + "loss": 0.5289290428161622, + "step": 2760 + }, + { + "epoch": 0.7453832489861582, + "grad_norm": 0.17643575370311737, + "learning_rate": 9.412198391420912e-06, + "loss": 0.548846435546875, + "step": 2780 + }, + { + "epoch": 0.7507457184033247, + "grad_norm": 0.23063655197620392, + "learning_rate": 9.371983914209116e-06, + "loss": 0.5502467155456543, + "step": 2800 + }, + { + "epoch": 0.7561081878204914, + "grad_norm": 0.14489713311195374, + "learning_rate": 9.33176943699732e-06, + "loss": 0.5205071449279786, + "step": 2820 + }, + { + "epoch": 0.7614706572376579, + "grad_norm": 0.15738680958747864, + "learning_rate": 9.291554959785523e-06, + "loss": 0.5463311195373535, + "step": 2840 + }, + { + "epoch": 0.7668331266548245, + "grad_norm": 0.1291189193725586, + "learning_rate": 9.251340482573727e-06, + "loss": 0.5183065414428711, + "step": 2860 + }, + { + "epoch": 0.7721955960719912, + "grad_norm": 0.14537270367145538, + "learning_rate": 9.21112600536193e-06, + "loss": 0.5544816493988037, + "step": 2880 + }, + { + "epoch": 0.7775580654891577, + "grad_norm": 0.13409097492694855, + "learning_rate": 9.170911528150134e-06, + "loss": 0.5107351303100586, + "step": 2900 + }, + { + "epoch": 0.7829205349063244, + "grad_norm": 0.2998020052909851, + "learning_rate": 9.130697050938338e-06, + "loss": 0.5310684680938721, + "step": 2920 + }, + { + "epoch": 0.788283004323491, + "grad_norm": 0.1838223934173584, + "learning_rate": 9.090482573726543e-06, + "loss": 0.5270499229431153, + "step": 2940 + }, + { + "epoch": 0.7936454737406575, + "grad_norm": 0.18618327379226685, + "learning_rate": 9.050268096514747e-06, + "loss": 0.5336289882659913, + "step": 2960 + }, + { + "epoch": 0.7990079431578242, + "grad_norm": 0.20681297779083252, + "learning_rate": 9.01005361930295e-06, + "loss": 0.508507251739502, + "step": 2980 + }, + { + "epoch": 0.8043704125749908, + "grad_norm": 0.24283935129642487, + "learning_rate": 8.969839142091154e-06, + "loss": 0.5339189052581788, + "step": 3000 + }, + { + "epoch": 0.8097328819921574, + "grad_norm": 0.21722275018692017, + "learning_rate": 8.929624664879358e-06, + "loss": 0.515669584274292, + "step": 3020 + }, + { + "epoch": 0.815095351409324, + "grad_norm": 0.14678969979286194, + "learning_rate": 8.889410187667562e-06, + "loss": 0.49359521865844724, + "step": 3040 + }, + { + "epoch": 0.8204578208264905, + "grad_norm": 0.16017946600914001, + "learning_rate": 8.849195710455765e-06, + "loss": 0.532757043838501, + "step": 3060 + }, + { + "epoch": 0.8258202902436572, + "grad_norm": 0.13103698194026947, + "learning_rate": 8.808981233243969e-06, + "loss": 0.5174227237701416, + "step": 3080 + }, + { + "epoch": 0.8311827596608238, + "grad_norm": 0.13764740526676178, + "learning_rate": 8.768766756032173e-06, + "loss": 0.5756002902984619, + "step": 3100 + }, + { + "epoch": 0.8365452290779904, + "grad_norm": 0.1956685334444046, + "learning_rate": 8.728552278820376e-06, + "loss": 0.5458150386810303, + "step": 3120 + }, + { + "epoch": 0.841907698495157, + "grad_norm": 0.14859093725681305, + "learning_rate": 8.68833780160858e-06, + "loss": 0.5232916831970215, + "step": 3140 + }, + { + "epoch": 0.8472701679123237, + "grad_norm": 0.14078572392463684, + "learning_rate": 8.648123324396784e-06, + "loss": 0.45665884017944336, + "step": 3160 + }, + { + "epoch": 0.8526326373294902, + "grad_norm": 0.10593896359205246, + "learning_rate": 8.607908847184988e-06, + "loss": 0.46901817321777345, + "step": 3180 + }, + { + "epoch": 0.8579951067466568, + "grad_norm": 0.19927014410495758, + "learning_rate": 8.567694369973191e-06, + "loss": 0.4962503910064697, + "step": 3200 + }, + { + "epoch": 0.8633575761638235, + "grad_norm": 0.1885233223438263, + "learning_rate": 8.527479892761395e-06, + "loss": 0.5428553581237793, + "step": 3220 + }, + { + "epoch": 0.86872004558099, + "grad_norm": 0.22774286568164825, + "learning_rate": 8.487265415549599e-06, + "loss": 0.5246198177337646, + "step": 3240 + }, + { + "epoch": 0.8740825149981567, + "grad_norm": 0.16228961944580078, + "learning_rate": 8.447050938337802e-06, + "loss": 0.5317719936370849, + "step": 3260 + }, + { + "epoch": 0.8794449844153233, + "grad_norm": 0.19011476635932922, + "learning_rate": 8.406836461126006e-06, + "loss": 0.5377527236938476, + "step": 3280 + }, + { + "epoch": 0.8848074538324898, + "grad_norm": 0.1937844604253769, + "learning_rate": 8.36662198391421e-06, + "loss": 0.5009727954864502, + "step": 3300 + }, + { + "epoch": 0.8901699232496565, + "grad_norm": 0.26362502574920654, + "learning_rate": 8.326407506702413e-06, + "loss": 0.5286832809448242, + "step": 3320 + }, + { + "epoch": 0.895532392666823, + "grad_norm": 0.15528951585292816, + "learning_rate": 8.286193029490617e-06, + "loss": 0.5699362754821777, + "step": 3340 + }, + { + "epoch": 0.9008948620839897, + "grad_norm": 0.19824309647083282, + "learning_rate": 8.24597855227882e-06, + "loss": 0.5417330265045166, + "step": 3360 + }, + { + "epoch": 0.9062573315011563, + "grad_norm": 0.17824552953243256, + "learning_rate": 8.205764075067025e-06, + "loss": 0.5166538238525391, + "step": 3380 + }, + { + "epoch": 0.9116198009183228, + "grad_norm": 0.1860542744398117, + "learning_rate": 8.165549597855228e-06, + "loss": 0.5525233745574951, + "step": 3400 + }, + { + "epoch": 0.9169822703354895, + "grad_norm": 0.22200629115104675, + "learning_rate": 8.125335120643432e-06, + "loss": 0.48862462043762206, + "step": 3420 + }, + { + "epoch": 0.9223447397526561, + "grad_norm": 0.21177783608436584, + "learning_rate": 8.085120643431636e-06, + "loss": 0.5362657070159912, + "step": 3440 + }, + { + "epoch": 0.9277072091698227, + "grad_norm": 0.1278514564037323, + "learning_rate": 8.04490616621984e-06, + "loss": 0.5472875595092773, + "step": 3460 + }, + { + "epoch": 0.9330696785869893, + "grad_norm": 0.1520422250032425, + "learning_rate": 8.004691689008043e-06, + "loss": 0.4906148910522461, + "step": 3480 + }, + { + "epoch": 0.9384321480041559, + "grad_norm": 0.1678784340620041, + "learning_rate": 7.964477211796247e-06, + "loss": 0.5190341949462891, + "step": 3500 + }, + { + "epoch": 0.9437946174213225, + "grad_norm": 0.2168162763118744, + "learning_rate": 7.92426273458445e-06, + "loss": 0.5007696151733398, + "step": 3520 + }, + { + "epoch": 0.9491570868384891, + "grad_norm": 0.18424147367477417, + "learning_rate": 7.884048257372654e-06, + "loss": 0.5395221710205078, + "step": 3540 + }, + { + "epoch": 0.9545195562556558, + "grad_norm": 0.17553555965423584, + "learning_rate": 7.843833780160858e-06, + "loss": 0.4716806888580322, + "step": 3560 + }, + { + "epoch": 0.9598820256728223, + "grad_norm": 0.15070843696594238, + "learning_rate": 7.803619302949062e-06, + "loss": 0.49967169761657715, + "step": 3580 + }, + { + "epoch": 0.9652444950899889, + "grad_norm": 0.172193244099617, + "learning_rate": 7.763404825737265e-06, + "loss": 0.495190954208374, + "step": 3600 + }, + { + "epoch": 0.9706069645071556, + "grad_norm": 0.15822157263755798, + "learning_rate": 7.723190348525469e-06, + "loss": 0.5322632789611816, + "step": 3620 + }, + { + "epoch": 0.9759694339243221, + "grad_norm": 0.19345910847187042, + "learning_rate": 7.682975871313673e-06, + "loss": 0.48404436111450194, + "step": 3640 + }, + { + "epoch": 0.9813319033414888, + "grad_norm": 0.17885969579219818, + "learning_rate": 7.642761394101876e-06, + "loss": 0.5166211128234863, + "step": 3660 + }, + { + "epoch": 0.9866943727586553, + "grad_norm": 0.15497833490371704, + "learning_rate": 7.60254691689008e-06, + "loss": 0.5560059547424316, + "step": 3680 + }, + { + "epoch": 0.992056842175822, + "grad_norm": 0.17155644297599792, + "learning_rate": 7.562332439678284e-06, + "loss": 0.529679822921753, + "step": 3700 + }, + { + "epoch": 0.9974193115929886, + "grad_norm": 0.18267494440078735, + "learning_rate": 7.522117962466487e-06, + "loss": 0.5055463790893555, + "step": 3720 + }, + { + "epoch": 1.0026812347085834, + "grad_norm": 0.1627507209777832, + "learning_rate": 7.481903485254692e-06, + "loss": 0.45867152214050294, + "step": 3740 + }, + { + "epoch": 1.00804370412575, + "grad_norm": 0.2230822890996933, + "learning_rate": 7.441689008042896e-06, + "loss": 0.4909696102142334, + "step": 3760 + }, + { + "epoch": 1.0134061735429165, + "grad_norm": 0.14418569207191467, + "learning_rate": 7.401474530831099e-06, + "loss": 0.4891301155090332, + "step": 3780 + }, + { + "epoch": 1.018768642960083, + "grad_norm": 0.2094171643257141, + "learning_rate": 7.361260053619303e-06, + "loss": 0.4919305324554443, + "step": 3800 + }, + { + "epoch": 1.0241311123772496, + "grad_norm": 0.16315558552742004, + "learning_rate": 7.321045576407507e-06, + "loss": 0.5338080406188965, + "step": 3820 + }, + { + "epoch": 1.0294935817944164, + "grad_norm": 0.20310278236865997, + "learning_rate": 7.2808310991957104e-06, + "loss": 0.4789735794067383, + "step": 3840 + }, + { + "epoch": 1.034856051211583, + "grad_norm": 0.13879640400409698, + "learning_rate": 7.240616621983915e-06, + "loss": 0.49851651191711427, + "step": 3860 + }, + { + "epoch": 1.0402185206287495, + "grad_norm": 0.1722245216369629, + "learning_rate": 7.200402144772119e-06, + "loss": 0.5306562900543212, + "step": 3880 + }, + { + "epoch": 1.045580990045916, + "grad_norm": 0.1506664901971817, + "learning_rate": 7.160187667560322e-06, + "loss": 0.45285625457763673, + "step": 3900 + }, + { + "epoch": 1.0509434594630827, + "grad_norm": 0.204021617770195, + "learning_rate": 7.119973190348526e-06, + "loss": 0.5161935329437256, + "step": 3920 + }, + { + "epoch": 1.0563059288802494, + "grad_norm": 0.20319899916648865, + "learning_rate": 7.07975871313673e-06, + "loss": 0.4824995040893555, + "step": 3940 + }, + { + "epoch": 1.061668398297416, + "grad_norm": 0.19432441890239716, + "learning_rate": 7.0395442359249335e-06, + "loss": 0.5660453796386719, + "step": 3960 + }, + { + "epoch": 1.0670308677145826, + "grad_norm": 0.2576168477535248, + "learning_rate": 6.999329758713137e-06, + "loss": 0.4815997123718262, + "step": 3980 + }, + { + "epoch": 1.0723933371317491, + "grad_norm": 0.27557438611984253, + "learning_rate": 6.959115281501341e-06, + "loss": 0.43416056632995603, + "step": 4000 + }, + { + "epoch": 1.0777558065489157, + "grad_norm": 0.17039135098457336, + "learning_rate": 6.9189008042895446e-06, + "loss": 0.4980440139770508, + "step": 4020 + }, + { + "epoch": 1.0831182759660825, + "grad_norm": 0.2580510675907135, + "learning_rate": 6.878686327077748e-06, + "loss": 0.5068618774414062, + "step": 4040 + }, + { + "epoch": 1.088480745383249, + "grad_norm": 0.14738141000270844, + "learning_rate": 6.838471849865952e-06, + "loss": 0.4890751361846924, + "step": 4060 + }, + { + "epoch": 1.0938432148004156, + "grad_norm": 0.2081380933523178, + "learning_rate": 6.798257372654156e-06, + "loss": 0.5679311275482177, + "step": 4080 + }, + { + "epoch": 1.0992056842175821, + "grad_norm": 0.17693300545215607, + "learning_rate": 6.758042895442359e-06, + "loss": 0.5189684391021728, + "step": 4100 + }, + { + "epoch": 1.104568153634749, + "grad_norm": 0.23674148321151733, + "learning_rate": 6.717828418230563e-06, + "loss": 0.48049330711364746, + "step": 4120 + }, + { + "epoch": 1.1099306230519155, + "grad_norm": 0.21366719901561737, + "learning_rate": 6.677613941018767e-06, + "loss": 0.4967336654663086, + "step": 4140 + }, + { + "epoch": 1.115293092469082, + "grad_norm": 0.19616496562957764, + "learning_rate": 6.6373994638069704e-06, + "loss": 0.46569108963012695, + "step": 4160 + }, + { + "epoch": 1.1206555618862486, + "grad_norm": 0.17559197545051575, + "learning_rate": 6.597184986595174e-06, + "loss": 0.49478998184204104, + "step": 4180 + }, + { + "epoch": 1.1260180313034152, + "grad_norm": 0.184451162815094, + "learning_rate": 6.556970509383378e-06, + "loss": 0.5000570774078369, + "step": 4200 + }, + { + "epoch": 1.131380500720582, + "grad_norm": 0.18627093732357025, + "learning_rate": 6.5167560321715815e-06, + "loss": 0.5214301586151123, + "step": 4220 + }, + { + "epoch": 1.1367429701377485, + "grad_norm": 0.2080899477005005, + "learning_rate": 6.476541554959785e-06, + "loss": 0.47851176261901857, + "step": 4240 + }, + { + "epoch": 1.142105439554915, + "grad_norm": 0.18619345128536224, + "learning_rate": 6.436327077747989e-06, + "loss": 0.5022239685058594, + "step": 4260 + }, + { + "epoch": 1.1474679089720816, + "grad_norm": 0.23693107068538666, + "learning_rate": 6.396112600536193e-06, + "loss": 0.5198223114013671, + "step": 4280 + }, + { + "epoch": 1.1528303783892482, + "grad_norm": 0.17998561263084412, + "learning_rate": 6.355898123324397e-06, + "loss": 0.5228567123413086, + "step": 4300 + }, + { + "epoch": 1.158192847806415, + "grad_norm": 0.2783758342266083, + "learning_rate": 6.315683646112601e-06, + "loss": 0.5318965435028076, + "step": 4320 + }, + { + "epoch": 1.1635553172235815, + "grad_norm": 0.19693782925605774, + "learning_rate": 6.2754691689008046e-06, + "loss": 0.48392295837402344, + "step": 4340 + }, + { + "epoch": 1.168917786640748, + "grad_norm": 0.15940269827842712, + "learning_rate": 6.235254691689008e-06, + "loss": 0.4617619514465332, + "step": 4360 + }, + { + "epoch": 1.1742802560579146, + "grad_norm": 0.24782665073871613, + "learning_rate": 6.195040214477212e-06, + "loss": 0.49810285568237306, + "step": 4380 + }, + { + "epoch": 1.1796427254750812, + "grad_norm": 0.1946037858724594, + "learning_rate": 6.154825737265416e-06, + "loss": 0.4826976776123047, + "step": 4400 + }, + { + "epoch": 1.185005194892248, + "grad_norm": 0.16667844355106354, + "learning_rate": 6.114611260053619e-06, + "loss": 0.5159809589385986, + "step": 4420 + }, + { + "epoch": 1.1903676643094145, + "grad_norm": 0.19206570088863373, + "learning_rate": 6.074396782841823e-06, + "loss": 0.47541089057922364, + "step": 4440 + }, + { + "epoch": 1.195730133726581, + "grad_norm": 0.17394617199897766, + "learning_rate": 6.034182305630027e-06, + "loss": 0.5470661640167236, + "step": 4460 + }, + { + "epoch": 1.2010926031437477, + "grad_norm": 0.210404634475708, + "learning_rate": 5.993967828418231e-06, + "loss": 0.5377882957458496, + "step": 4480 + }, + { + "epoch": 1.2064550725609142, + "grad_norm": 0.18084648251533508, + "learning_rate": 5.953753351206435e-06, + "loss": 0.5037185192108155, + "step": 4500 + }, + { + "epoch": 1.211817541978081, + "grad_norm": 0.23707027733325958, + "learning_rate": 5.913538873994639e-06, + "loss": 0.4822190284729004, + "step": 4520 + }, + { + "epoch": 1.2171800113952476, + "grad_norm": 0.16474473476409912, + "learning_rate": 5.873324396782842e-06, + "loss": 0.46645288467407225, + "step": 4540 + }, + { + "epoch": 1.2225424808124141, + "grad_norm": 0.2142348438501358, + "learning_rate": 5.833109919571046e-06, + "loss": 0.5255855560302735, + "step": 4560 + }, + { + "epoch": 1.2279049502295807, + "grad_norm": 0.2531765103340149, + "learning_rate": 5.79289544235925e-06, + "loss": 0.507044792175293, + "step": 4580 + }, + { + "epoch": 1.2332674196467472, + "grad_norm": 0.2553550899028778, + "learning_rate": 5.7526809651474535e-06, + "loss": 0.4767824649810791, + "step": 4600 + }, + { + "epoch": 1.238629889063914, + "grad_norm": 0.14484412968158722, + "learning_rate": 5.712466487935657e-06, + "loss": 0.4675601005554199, + "step": 4620 + }, + { + "epoch": 1.2439923584810806, + "grad_norm": 0.14328251779079437, + "learning_rate": 5.672252010723861e-06, + "loss": 0.4956005573272705, + "step": 4640 + }, + { + "epoch": 1.2493548278982471, + "grad_norm": 0.1739245355129242, + "learning_rate": 5.632037533512065e-06, + "loss": 0.48583345413208007, + "step": 4660 + }, + { + "epoch": 1.2547172973154137, + "grad_norm": 0.21294184029102325, + "learning_rate": 5.591823056300268e-06, + "loss": 0.520921277999878, + "step": 4680 + }, + { + "epoch": 1.2600797667325803, + "grad_norm": 0.25132355093955994, + "learning_rate": 5.551608579088472e-06, + "loss": 0.5295385837554931, + "step": 4700 + }, + { + "epoch": 1.265442236149747, + "grad_norm": 0.18603841960430145, + "learning_rate": 5.511394101876676e-06, + "loss": 0.47570199966430665, + "step": 4720 + }, + { + "epoch": 1.2708047055669136, + "grad_norm": 0.19883134961128235, + "learning_rate": 5.471179624664879e-06, + "loss": 0.5016080379486084, + "step": 4740 + }, + { + "epoch": 1.2761671749840802, + "grad_norm": 0.19640181958675385, + "learning_rate": 5.430965147453083e-06, + "loss": 0.4999081134796143, + "step": 4760 + }, + { + "epoch": 1.2815296444012467, + "grad_norm": 0.2584764361381531, + "learning_rate": 5.390750670241287e-06, + "loss": 0.4780082702636719, + "step": 4780 + }, + { + "epoch": 1.2868921138184133, + "grad_norm": 0.2925741374492645, + "learning_rate": 5.3505361930294905e-06, + "loss": 0.5131395816802978, + "step": 4800 + }, + { + "epoch": 1.29225458323558, + "grad_norm": 0.18971531093120575, + "learning_rate": 5.310321715817694e-06, + "loss": 0.455674409866333, + "step": 4820 + }, + { + "epoch": 1.2976170526527466, + "grad_norm": 0.16778405010700226, + "learning_rate": 5.270107238605898e-06, + "loss": 0.5070962905883789, + "step": 4840 + }, + { + "epoch": 1.3029795220699132, + "grad_norm": 0.30026957392692566, + "learning_rate": 5.2298927613941016e-06, + "loss": 0.5120027542114258, + "step": 4860 + }, + { + "epoch": 1.3083419914870797, + "grad_norm": 0.17846634984016418, + "learning_rate": 5.189678284182305e-06, + "loss": 0.5114477157592774, + "step": 4880 + }, + { + "epoch": 1.3137044609042463, + "grad_norm": 0.1962418258190155, + "learning_rate": 5.149463806970509e-06, + "loss": 0.5043613910675049, + "step": 4900 + }, + { + "epoch": 1.319066930321413, + "grad_norm": 0.18446756899356842, + "learning_rate": 5.1092493297587135e-06, + "loss": 0.5396455287933349, + "step": 4920 + }, + { + "epoch": 1.3244293997385796, + "grad_norm": 0.20886844396591187, + "learning_rate": 5.069034852546917e-06, + "loss": 0.4879767417907715, + "step": 4940 + }, + { + "epoch": 1.3297918691557462, + "grad_norm": 0.16687901318073273, + "learning_rate": 5.028820375335121e-06, + "loss": 0.5014327049255372, + "step": 4960 + }, + { + "epoch": 1.3351543385729128, + "grad_norm": 0.19595153629779816, + "learning_rate": 4.988605898123325e-06, + "loss": 0.5375277996063232, + "step": 4980 + }, + { + "epoch": 1.3405168079900793, + "grad_norm": 0.2372344732284546, + "learning_rate": 4.948391420911528e-06, + "loss": 0.5020076274871826, + "step": 5000 + }, + { + "epoch": 1.345879277407246, + "grad_norm": 0.21030014753341675, + "learning_rate": 4.908176943699732e-06, + "loss": 0.5111066818237304, + "step": 5020 + }, + { + "epoch": 1.3512417468244127, + "grad_norm": 0.1866692751646042, + "learning_rate": 4.867962466487936e-06, + "loss": 0.4515383720397949, + "step": 5040 + }, + { + "epoch": 1.3566042162415792, + "grad_norm": 0.22531798481941223, + "learning_rate": 4.827747989276139e-06, + "loss": 0.4757690906524658, + "step": 5060 + }, + { + "epoch": 1.3619666856587458, + "grad_norm": 0.15868768095970154, + "learning_rate": 4.787533512064343e-06, + "loss": 0.45842318534851073, + "step": 5080 + }, + { + "epoch": 1.3673291550759124, + "grad_norm": 0.24528546631336212, + "learning_rate": 4.747319034852547e-06, + "loss": 0.47269258499145506, + "step": 5100 + }, + { + "epoch": 1.3726916244930791, + "grad_norm": 0.17387732863426208, + "learning_rate": 4.707104557640751e-06, + "loss": 0.5103805065155029, + "step": 5120 + }, + { + "epoch": 1.3780540939102457, + "grad_norm": 0.20686905086040497, + "learning_rate": 4.666890080428955e-06, + "loss": 0.5135180950164795, + "step": 5140 + }, + { + "epoch": 1.3834165633274123, + "grad_norm": 0.19599783420562744, + "learning_rate": 4.626675603217159e-06, + "loss": 0.5045839786529541, + "step": 5160 + }, + { + "epoch": 1.3887790327445788, + "grad_norm": 0.2585010528564453, + "learning_rate": 4.586461126005362e-06, + "loss": 0.45903496742248534, + "step": 5180 + }, + { + "epoch": 1.3941415021617454, + "grad_norm": 0.1688319593667984, + "learning_rate": 4.546246648793566e-06, + "loss": 0.5017509937286377, + "step": 5200 + }, + { + "epoch": 1.3995039715789122, + "grad_norm": 0.21520815789699554, + "learning_rate": 4.50603217158177e-06, + "loss": 0.48459539413452146, + "step": 5220 + }, + { + "epoch": 1.4048664409960787, + "grad_norm": 0.20514647662639618, + "learning_rate": 4.4658176943699735e-06, + "loss": 0.5073423862457276, + "step": 5240 + }, + { + "epoch": 1.4102289104132453, + "grad_norm": 0.21835413575172424, + "learning_rate": 4.425603217158177e-06, + "loss": 0.5290310382843018, + "step": 5260 + }, + { + "epoch": 1.4155913798304118, + "grad_norm": 0.28042587637901306, + "learning_rate": 4.385388739946381e-06, + "loss": 0.4823312759399414, + "step": 5280 + }, + { + "epoch": 1.4209538492475784, + "grad_norm": 0.18959026038646698, + "learning_rate": 4.345174262734585e-06, + "loss": 0.4921241760253906, + "step": 5300 + }, + { + "epoch": 1.4263163186647452, + "grad_norm": 0.18584316968917847, + "learning_rate": 4.304959785522788e-06, + "loss": 0.4892130374908447, + "step": 5320 + }, + { + "epoch": 1.4316787880819117, + "grad_norm": 0.17588038742542267, + "learning_rate": 4.264745308310992e-06, + "loss": 0.4822041988372803, + "step": 5340 + }, + { + "epoch": 1.4370412574990783, + "grad_norm": 0.18146033585071564, + "learning_rate": 4.224530831099196e-06, + "loss": 0.5084807395935058, + "step": 5360 + }, + { + "epoch": 1.4424037269162449, + "grad_norm": 0.2251797467470169, + "learning_rate": 4.184316353887399e-06, + "loss": 0.5146170139312745, + "step": 5380 + }, + { + "epoch": 1.4477661963334114, + "grad_norm": 0.18744796514511108, + "learning_rate": 4.144101876675603e-06, + "loss": 0.5189927577972412, + "step": 5400 + }, + { + "epoch": 1.4531286657505782, + "grad_norm": 0.25737133622169495, + "learning_rate": 4.103887399463807e-06, + "loss": 0.4891658782958984, + "step": 5420 + }, + { + "epoch": 1.4584911351677448, + "grad_norm": 0.20580479502677917, + "learning_rate": 4.0636729222520105e-06, + "loss": 0.4953591823577881, + "step": 5440 + }, + { + "epoch": 1.4638536045849113, + "grad_norm": 0.2351546287536621, + "learning_rate": 4.023458445040214e-06, + "loss": 0.5025320053100586, + "step": 5460 + }, + { + "epoch": 1.4692160740020779, + "grad_norm": 0.1819481998682022, + "learning_rate": 3.983243967828418e-06, + "loss": 0.47151756286621094, + "step": 5480 + }, + { + "epoch": 1.4745785434192444, + "grad_norm": 0.20772472023963928, + "learning_rate": 3.943029490616622e-06, + "loss": 0.4678915023803711, + "step": 5500 + }, + { + "epoch": 1.4799410128364112, + "grad_norm": 0.2203037440776825, + "learning_rate": 3.902815013404825e-06, + "loss": 0.46007452011108396, + "step": 5520 + }, + { + "epoch": 1.4853034822535778, + "grad_norm": 0.15371400117874146, + "learning_rate": 3.86260053619303e-06, + "loss": 0.44407024383544924, + "step": 5540 + }, + { + "epoch": 1.4906659516707443, + "grad_norm": 0.2276080846786499, + "learning_rate": 3.8223860589812335e-06, + "loss": 0.4730556488037109, + "step": 5560 + }, + { + "epoch": 1.4960284210879111, + "grad_norm": 0.24482466280460358, + "learning_rate": 3.7821715817694376e-06, + "loss": 0.5073911666870117, + "step": 5580 + }, + { + "epoch": 1.5013908905050775, + "grad_norm": 0.20438458025455475, + "learning_rate": 3.741957104557641e-06, + "loss": 0.46701641082763673, + "step": 5600 + }, + { + "epoch": 1.5067533599222442, + "grad_norm": 0.19854313135147095, + "learning_rate": 3.7017426273458446e-06, + "loss": 0.46309399604797363, + "step": 5620 + }, + { + "epoch": 1.5121158293394108, + "grad_norm": 0.18356069922447205, + "learning_rate": 3.6615281501340483e-06, + "loss": 0.503613805770874, + "step": 5640 + }, + { + "epoch": 1.5174782987565774, + "grad_norm": 0.2009744495153427, + "learning_rate": 3.621313672922252e-06, + "loss": 0.4765054225921631, + "step": 5660 + }, + { + "epoch": 1.5228407681737441, + "grad_norm": 0.3058745563030243, + "learning_rate": 3.5810991957104557e-06, + "loss": 0.5179148197174073, + "step": 5680 + }, + { + "epoch": 1.5282032375909105, + "grad_norm": 0.17671597003936768, + "learning_rate": 3.54088471849866e-06, + "loss": 0.45907344818115237, + "step": 5700 + }, + { + "epoch": 1.5335657070080773, + "grad_norm": 0.22209160029888153, + "learning_rate": 3.5006702412868635e-06, + "loss": 0.49304862022399903, + "step": 5720 + }, + { + "epoch": 1.5389281764252438, + "grad_norm": 0.21018914878368378, + "learning_rate": 3.4604557640750672e-06, + "loss": 0.5536758422851562, + "step": 5740 + }, + { + "epoch": 1.5442906458424104, + "grad_norm": 0.14339996874332428, + "learning_rate": 3.420241286863271e-06, + "loss": 0.48726091384887693, + "step": 5760 + }, + { + "epoch": 1.5496531152595772, + "grad_norm": 0.11419746279716492, + "learning_rate": 3.3800268096514746e-06, + "loss": 0.4514151573181152, + "step": 5780 + }, + { + "epoch": 1.5550155846767435, + "grad_norm": 0.18168962001800537, + "learning_rate": 3.3398123324396783e-06, + "loss": 0.5279990196228027, + "step": 5800 + }, + { + "epoch": 1.5603780540939103, + "grad_norm": 0.24244488775730133, + "learning_rate": 3.299597855227882e-06, + "loss": 0.49297361373901366, + "step": 5820 + }, + { + "epoch": 1.5657405235110768, + "grad_norm": 0.2017296999692917, + "learning_rate": 3.2593833780160857e-06, + "loss": 0.49305019378662107, + "step": 5840 + }, + { + "epoch": 1.5711029929282434, + "grad_norm": 0.22592377662658691, + "learning_rate": 3.2191689008042894e-06, + "loss": 0.4862989902496338, + "step": 5860 + }, + { + "epoch": 1.5764654623454102, + "grad_norm": 0.24772357940673828, + "learning_rate": 3.1789544235924935e-06, + "loss": 0.45182647705078127, + "step": 5880 + }, + { + "epoch": 1.5818279317625765, + "grad_norm": 0.20607218146324158, + "learning_rate": 3.1387399463806972e-06, + "loss": 0.48905248641967775, + "step": 5900 + }, + { + "epoch": 1.5871904011797433, + "grad_norm": 0.1931353509426117, + "learning_rate": 3.098525469168901e-06, + "loss": 0.5307461261749268, + "step": 5920 + }, + { + "epoch": 1.5925528705969099, + "grad_norm": 0.16020581126213074, + "learning_rate": 3.0583109919571046e-06, + "loss": 0.4672811985015869, + "step": 5940 + }, + { + "epoch": 1.5979153400140764, + "grad_norm": 0.23668015003204346, + "learning_rate": 3.0180965147453083e-06, + "loss": 0.5272688865661621, + "step": 5960 + }, + { + "epoch": 1.6032778094312432, + "grad_norm": 0.1916576772928238, + "learning_rate": 2.977882037533512e-06, + "loss": 0.4859332084655762, + "step": 5980 + }, + { + "epoch": 1.6086402788484095, + "grad_norm": 0.23635101318359375, + "learning_rate": 2.9376675603217157e-06, + "loss": 0.5418910980224609, + "step": 6000 + }, + { + "epoch": 1.6140027482655763, + "grad_norm": 0.2404562532901764, + "learning_rate": 2.89745308310992e-06, + "loss": 0.5449445247650146, + "step": 6020 + }, + { + "epoch": 1.6193652176827429, + "grad_norm": 0.20147347450256348, + "learning_rate": 2.8572386058981235e-06, + "loss": 0.4737790584564209, + "step": 6040 + }, + { + "epoch": 1.6247276870999094, + "grad_norm": 0.2455863654613495, + "learning_rate": 2.8170241286863272e-06, + "loss": 0.4722298145294189, + "step": 6060 + }, + { + "epoch": 1.6300901565170762, + "grad_norm": 0.22172148525714874, + "learning_rate": 2.776809651474531e-06, + "loss": 0.5120372295379638, + "step": 6080 + }, + { + "epoch": 1.6354526259342426, + "grad_norm": 0.3848462700843811, + "learning_rate": 2.7365951742627346e-06, + "loss": 0.5152206897735596, + "step": 6100 + }, + { + "epoch": 1.6408150953514093, + "grad_norm": 0.19071047008037567, + "learning_rate": 2.6963806970509383e-06, + "loss": 0.4757692813873291, + "step": 6120 + }, + { + "epoch": 1.646177564768576, + "grad_norm": 0.20568661391735077, + "learning_rate": 2.656166219839142e-06, + "loss": 0.475917387008667, + "step": 6140 + }, + { + "epoch": 1.6515400341857425, + "grad_norm": 0.11777322739362717, + "learning_rate": 2.6159517426273457e-06, + "loss": 0.5161296367645264, + "step": 6160 + }, + { + "epoch": 1.6569025036029092, + "grad_norm": 0.1700555831193924, + "learning_rate": 2.5757372654155494e-06, + "loss": 0.4715432167053223, + "step": 6180 + }, + { + "epoch": 1.6622649730200756, + "grad_norm": 0.18927083909511566, + "learning_rate": 2.5355227882037535e-06, + "loss": 0.49937710762023924, + "step": 6200 + }, + { + "epoch": 1.6676274424372424, + "grad_norm": 0.22097784280776978, + "learning_rate": 2.4953083109919572e-06, + "loss": 0.43366107940673826, + "step": 6220 + }, + { + "epoch": 1.672989911854409, + "grad_norm": 0.2299281805753708, + "learning_rate": 2.455093833780161e-06, + "loss": 0.5145821094512939, + "step": 6240 + }, + { + "epoch": 1.6783523812715755, + "grad_norm": 0.2384844720363617, + "learning_rate": 2.4148793565683646e-06, + "loss": 0.459308385848999, + "step": 6260 + }, + { + "epoch": 1.6837148506887423, + "grad_norm": 0.24471035599708557, + "learning_rate": 2.3746648793565683e-06, + "loss": 0.4676504611968994, + "step": 6280 + }, + { + "epoch": 1.6890773201059086, + "grad_norm": 0.24419866502285004, + "learning_rate": 2.334450402144772e-06, + "loss": 0.4745138168334961, + "step": 6300 + }, + { + "epoch": 1.6944397895230754, + "grad_norm": 0.15896575152873993, + "learning_rate": 2.294235924932976e-06, + "loss": 0.5073649883270264, + "step": 6320 + }, + { + "epoch": 1.699802258940242, + "grad_norm": 0.26504868268966675, + "learning_rate": 2.25402144772118e-06, + "loss": 0.4534353733062744, + "step": 6340 + }, + { + "epoch": 1.7051647283574085, + "grad_norm": 0.2461850792169571, + "learning_rate": 2.2138069705093836e-06, + "loss": 0.4862947940826416, + "step": 6360 + }, + { + "epoch": 1.7105271977745753, + "grad_norm": 0.17332817614078522, + "learning_rate": 2.1735924932975873e-06, + "loss": 0.5049370765686035, + "step": 6380 + }, + { + "epoch": 1.7158896671917419, + "grad_norm": 0.19762548804283142, + "learning_rate": 2.133378016085791e-06, + "loss": 0.5272616386413574, + "step": 6400 + }, + { + "epoch": 1.7212521366089084, + "grad_norm": 0.23265399038791656, + "learning_rate": 2.0931635388739946e-06, + "loss": 0.47600841522216797, + "step": 6420 + }, + { + "epoch": 1.726614606026075, + "grad_norm": 0.20868578553199768, + "learning_rate": 2.0529490616621983e-06, + "loss": 0.5027226448059082, + "step": 6440 + }, + { + "epoch": 1.7319770754432415, + "grad_norm": 0.2851981520652771, + "learning_rate": 2.012734584450402e-06, + "loss": 0.5288124561309815, + "step": 6460 + }, + { + "epoch": 1.7373395448604083, + "grad_norm": 0.20086587965488434, + "learning_rate": 1.9725201072386057e-06, + "loss": 0.4625516891479492, + "step": 6480 + }, + { + "epoch": 1.7427020142775749, + "grad_norm": 0.24060192704200745, + "learning_rate": 1.93230563002681e-06, + "loss": 0.4843903541564941, + "step": 6500 + }, + { + "epoch": 1.7480644836947414, + "grad_norm": 0.33561915159225464, + "learning_rate": 1.8920911528150133e-06, + "loss": 0.4823720932006836, + "step": 6520 + }, + { + "epoch": 1.753426953111908, + "grad_norm": 0.2510465383529663, + "learning_rate": 1.851876675603217e-06, + "loss": 0.46517143249511717, + "step": 6540 + }, + { + "epoch": 1.7587894225290746, + "grad_norm": 0.2631177604198456, + "learning_rate": 1.811662198391421e-06, + "loss": 0.5004732131958007, + "step": 6560 + }, + { + "epoch": 1.7641518919462413, + "grad_norm": 0.3493230640888214, + "learning_rate": 1.7714477211796249e-06, + "loss": 0.523811674118042, + "step": 6580 + }, + { + "epoch": 1.769514361363408, + "grad_norm": 0.1742691546678543, + "learning_rate": 1.7312332439678286e-06, + "loss": 0.5276295661926269, + "step": 6600 + }, + { + "epoch": 1.7748768307805745, + "grad_norm": 0.16134823858737946, + "learning_rate": 1.6910187667560323e-06, + "loss": 0.5352637290954589, + "step": 6620 + }, + { + "epoch": 1.780239300197741, + "grad_norm": 0.20977018773555756, + "learning_rate": 1.650804289544236e-06, + "loss": 0.4955774784088135, + "step": 6640 + }, + { + "epoch": 1.7856017696149076, + "grad_norm": 0.20511005818843842, + "learning_rate": 1.6105898123324397e-06, + "loss": 0.48643174171447756, + "step": 6660 + }, + { + "epoch": 1.7909642390320744, + "grad_norm": 0.23870044946670532, + "learning_rate": 1.5703753351206434e-06, + "loss": 0.4673162460327148, + "step": 6680 + }, + { + "epoch": 1.796326708449241, + "grad_norm": 0.21660065650939941, + "learning_rate": 1.5301608579088473e-06, + "loss": 0.5381903648376465, + "step": 6700 + }, + { + "epoch": 1.8016891778664075, + "grad_norm": 0.26977139711380005, + "learning_rate": 1.489946380697051e-06, + "loss": 0.42094998359680175, + "step": 6720 + }, + { + "epoch": 1.807051647283574, + "grad_norm": 0.2088550478219986, + "learning_rate": 1.4497319034852549e-06, + "loss": 0.49211792945861815, + "step": 6740 + }, + { + "epoch": 1.8124141167007406, + "grad_norm": 0.18141885101795197, + "learning_rate": 1.4095174262734586e-06, + "loss": 0.46572179794311525, + "step": 6760 + }, + { + "epoch": 1.8177765861179074, + "grad_norm": 0.2200685739517212, + "learning_rate": 1.3693029490616623e-06, + "loss": 0.4996177196502686, + "step": 6780 + }, + { + "epoch": 1.823139055535074, + "grad_norm": 0.19545452296733856, + "learning_rate": 1.329088471849866e-06, + "loss": 0.4731945514678955, + "step": 6800 + }, + { + "epoch": 1.8285015249522405, + "grad_norm": 0.2239731252193451, + "learning_rate": 1.2888739946380697e-06, + "loss": 0.47544050216674805, + "step": 6820 + }, + { + "epoch": 1.833863994369407, + "grad_norm": 0.22336581349372864, + "learning_rate": 1.2486595174262734e-06, + "loss": 0.47878737449645997, + "step": 6840 + }, + { + "epoch": 1.8392264637865736, + "grad_norm": 0.20921571552753448, + "learning_rate": 1.2084450402144773e-06, + "loss": 0.41347403526306153, + "step": 6860 + }, + { + "epoch": 1.8445889332037404, + "grad_norm": 0.1577194333076477, + "learning_rate": 1.168230563002681e-06, + "loss": 0.5464958667755127, + "step": 6880 + }, + { + "epoch": 1.849951402620907, + "grad_norm": 0.1477355808019638, + "learning_rate": 1.1280160857908849e-06, + "loss": 0.48548617362976076, + "step": 6900 + }, + { + "epoch": 1.8553138720380735, + "grad_norm": 0.22352682054042816, + "learning_rate": 1.0878016085790886e-06, + "loss": 0.4518588542938232, + "step": 6920 + }, + { + "epoch": 1.8606763414552403, + "grad_norm": 0.19822706282138824, + "learning_rate": 1.0475871313672923e-06, + "loss": 0.4190972805023193, + "step": 6940 + }, + { + "epoch": 1.8660388108724066, + "grad_norm": 0.20670010149478912, + "learning_rate": 1.007372654155496e-06, + "loss": 0.5045090675354004, + "step": 6960 + }, + { + "epoch": 1.8714012802895734, + "grad_norm": 0.2154514342546463, + "learning_rate": 9.671581769436997e-07, + "loss": 0.4472982883453369, + "step": 6980 + }, + { + "epoch": 1.87676374970674, + "grad_norm": 0.19451302289962769, + "learning_rate": 9.269436997319035e-07, + "loss": 0.4328409194946289, + "step": 7000 + }, + { + "epoch": 1.8821262191239065, + "grad_norm": 0.20980985462665558, + "learning_rate": 8.867292225201073e-07, + "loss": 0.43312845230102537, + "step": 7020 + }, + { + "epoch": 1.8874886885410733, + "grad_norm": 0.19927652180194855, + "learning_rate": 8.46514745308311e-07, + "loss": 0.5255829811096191, + "step": 7040 + }, + { + "epoch": 1.8928511579582397, + "grad_norm": 0.2869090437889099, + "learning_rate": 8.063002680965148e-07, + "loss": 0.5301108360290527, + "step": 7060 + }, + { + "epoch": 1.8982136273754064, + "grad_norm": 0.16662724316120148, + "learning_rate": 7.660857908847185e-07, + "loss": 0.48042588233947753, + "step": 7080 + }, + { + "epoch": 1.903576096792573, + "grad_norm": 0.21045279502868652, + "learning_rate": 7.258713136729223e-07, + "loss": 0.4943391799926758, + "step": 7100 + }, + { + "epoch": 1.9089385662097396, + "grad_norm": 0.18638195097446442, + "learning_rate": 6.85656836461126e-07, + "loss": 0.48406500816345216, + "step": 7120 + }, + { + "epoch": 1.9143010356269063, + "grad_norm": 0.15428832173347473, + "learning_rate": 6.454423592493298e-07, + "loss": 0.5084923267364502, + "step": 7140 + }, + { + "epoch": 1.9196635050440727, + "grad_norm": 0.2415294051170349, + "learning_rate": 6.052278820375336e-07, + "loss": 0.5026498317718506, + "step": 7160 + }, + { + "epoch": 1.9250259744612395, + "grad_norm": 0.23021087050437927, + "learning_rate": 5.650134048257373e-07, + "loss": 0.5294596195220947, + "step": 7180 + }, + { + "epoch": 1.930388443878406, + "grad_norm": 0.21689893305301666, + "learning_rate": 5.24798927613941e-07, + "loss": 0.4569683074951172, + "step": 7200 + }, + { + "epoch": 1.9357509132955726, + "grad_norm": 0.25458022952079773, + "learning_rate": 4.845844504021448e-07, + "loss": 0.4905412197113037, + "step": 7220 + }, + { + "epoch": 1.9411133827127394, + "grad_norm": 0.20990079641342163, + "learning_rate": 4.4436997319034854e-07, + "loss": 0.49599390029907225, + "step": 7240 + }, + { + "epoch": 1.9464758521299057, + "grad_norm": 0.2381497025489807, + "learning_rate": 4.041554959785523e-07, + "loss": 0.5149903774261475, + "step": 7260 + }, + { + "epoch": 1.9518383215470725, + "grad_norm": 0.29006749391555786, + "learning_rate": 3.6394101876675604e-07, + "loss": 0.5063531875610352, + "step": 7280 + }, + { + "epoch": 1.957200790964239, + "grad_norm": 0.20159471035003662, + "learning_rate": 3.237265415549598e-07, + "loss": 0.5062472343444824, + "step": 7300 + }, + { + "epoch": 1.9625632603814056, + "grad_norm": 0.21182367205619812, + "learning_rate": 2.8351206434316354e-07, + "loss": 0.49964118003845215, + "step": 7320 + }, + { + "epoch": 1.9679257297985724, + "grad_norm": 0.28606927394866943, + "learning_rate": 2.432975871313673e-07, + "loss": 0.48647170066833495, + "step": 7340 + }, + { + "epoch": 1.9732881992157387, + "grad_norm": 0.2779375910758972, + "learning_rate": 2.0308310991957104e-07, + "loss": 0.4816920280456543, + "step": 7360 + }, + { + "epoch": 1.9786506686329055, + "grad_norm": 0.20412451028823853, + "learning_rate": 1.628686327077748e-07, + "loss": 0.5177248477935791, + "step": 7380 + }, + { + "epoch": 1.984013138050072, + "grad_norm": 0.19861914217472076, + "learning_rate": 1.2265415549597854e-07, + "loss": 0.5265318870544433, + "step": 7400 + }, + { + "epoch": 1.9893756074672386, + "grad_norm": 0.1868918091058731, + "learning_rate": 8.24396782841823e-08, + "loss": 0.45395827293395996, + "step": 7420 + }, + { + "epoch": 1.9947380768844054, + "grad_norm": 0.17822134494781494, + "learning_rate": 4.2225201072386056e-08, + "loss": 0.4699725151062012, + "step": 7440 + }, + { + "epoch": 2.0, + "grad_norm": 0.2196798026561737, + "learning_rate": 2.0107238605898125e-09, + "loss": 0.46166625022888186, + "step": 7460 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 9.191746827630674e+17, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-7460/training_args.bin b/checkpoint-7460/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-7460/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/checkpoint-800/README.md b/checkpoint-800/README.md new file mode 100644 index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6 --- /dev/null +++ b/checkpoint-800/README.md @@ -0,0 +1,206 @@ +--- +base_model: Qwen/Qwen2.5-14B +library_name: peft +pipeline_tag: text-generation +tags: +- base_model:adapter:Qwen/Qwen2.5-14B +- lora +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.18.1 \ No newline at end of file diff --git a/checkpoint-800/adapter_config.json b/checkpoint-800/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48 --- /dev/null +++ b/checkpoint-800/adapter_config.json @@ -0,0 +1,41 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": null, + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": false, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 32, + "lora_bias": false, + "lora_dropout": 0.05, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.18.1", + "qalora_group_size": 16, + "r": 16, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "q_proj" + ], + "target_parameters": null, + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/checkpoint-800/adapter_model.safetensors b/checkpoint-800/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..428fbbaae23118003b5f4ebfa7a79f2ab552e0c8 --- /dev/null +++ b/checkpoint-800/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b62aa4030982e8e3169a6adbec2f7a45ebe68eea7f45e8b730b8656e9414ac52 +size 50360752 diff --git a/checkpoint-800/chat_template.jinja b/checkpoint-800/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1 --- /dev/null +++ b/checkpoint-800/chat_template.jinja @@ -0,0 +1,54 @@ +{%- if tools %} + {{- '<|im_start|>system\n' }} + {%- if messages[0]['role'] == 'system' %} + {{- messages[0]['content'] }} + {%- else %} + {{- 'You are a helpful assistant.' }} + {%- endif %} + {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }} +{%- else %} + {%- if messages[0]['role'] == 'system' %} + {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }} + {%- else %} + {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- for message in messages %} + {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %} + {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {{- '<|im_start|>' + message.role }} + {%- if message.content %} + {{- '\n' + message.content }} + {%- endif %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {{- '\n\n{"name": "' }} + {{- tool_call.name }} + {{- '", "arguments": ' }} + {{- tool_call.arguments | tojson }} + {{- '}\n' }} + {%- endfor %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- message.content }} + {{- '\n' }} + {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} +{%- endif %} diff --git a/checkpoint-800/optimizer.pt b/checkpoint-800/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..47d0cc019ff07ea57013a8ce804948317ed6166a --- /dev/null +++ b/checkpoint-800/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a60aceea63941b6f07d7d1b9d66cfdfe24c0bd021d2f9f656d097cc8b57e327b +size 100828235 diff --git a/checkpoint-800/rng_state.pth b/checkpoint-800/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a76e5716169bf8900f8577fd27094187888a25e6 --- /dev/null +++ b/checkpoint-800/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:63dd7726a052e5bc7ad80bc87b567a2ece2f71b63cea73b5c0f39c143f56dfe5 +size 14645 diff --git a/checkpoint-800/scheduler.pt b/checkpoint-800/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bbdce98612b26bd9935d387661e0d77ab2b596bc --- /dev/null +++ b/checkpoint-800/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c770d5e2f41bbfe7a718e461fc6c35726d183615e87d7855afe86b2ceea37a85 +size 1465 diff --git a/checkpoint-800/tokenizer.json b/checkpoint-800/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/checkpoint-800/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/checkpoint-800/tokenizer_config.json b/checkpoint-800/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/checkpoint-800/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/checkpoint-800/trainer_state.json b/checkpoint-800/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..111742b7c8a060594898a67337614563824e088e --- /dev/null +++ b/checkpoint-800/trainer_state.json @@ -0,0 +1,314 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.2144987766866642, + "eval_steps": 500, + "global_step": 800, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.005362469417166605, + "grad_norm": 0.050072263926267624, + "learning_rate": 1.4961796246648793e-05, + "loss": 1.0673207283020019, + "step": 20 + }, + { + "epoch": 0.01072493883433321, + "grad_norm": 0.06825340539216995, + "learning_rate": 1.4921581769436997e-05, + "loss": 0.9185627937316895, + "step": 40 + }, + { + "epoch": 0.016087408251499815, + "grad_norm": 0.06827432662248611, + "learning_rate": 1.48813672922252e-05, + "loss": 0.7999343872070312, + "step": 60 + }, + { + "epoch": 0.02144987766866642, + "grad_norm": 0.05807405710220337, + "learning_rate": 1.4841152815013404e-05, + "loss": 0.7322770595550537, + "step": 80 + }, + { + "epoch": 0.026812347085833025, + "grad_norm": 0.06654328852891922, + "learning_rate": 1.4800938337801608e-05, + "loss": 0.7097890377044678, + "step": 100 + }, + { + "epoch": 0.03217481650299963, + "grad_norm": 0.09104783087968826, + "learning_rate": 1.4760723860589812e-05, + "loss": 0.6513629913330078, + "step": 120 + }, + { + "epoch": 0.03753728592016624, + "grad_norm": 0.10718850791454315, + "learning_rate": 1.4720509383378015e-05, + "loss": 0.678717851638794, + "step": 140 + }, + { + "epoch": 0.04289975533733284, + "grad_norm": 0.09187154471874237, + "learning_rate": 1.4680294906166219e-05, + "loss": 0.647278118133545, + "step": 160 + }, + { + "epoch": 0.04826222475449945, + "grad_norm": 0.07148946076631546, + "learning_rate": 1.4640080428954423e-05, + "loss": 0.6737877368927002, + "step": 180 + }, + { + "epoch": 0.05362469417166605, + "grad_norm": 0.08909227699041367, + "learning_rate": 1.4599865951742626e-05, + "loss": 0.6373191356658936, + "step": 200 + }, + { + "epoch": 0.05898716358883266, + "grad_norm": 0.07850278168916702, + "learning_rate": 1.455965147453083e-05, + "loss": 0.6020126819610596, + "step": 220 + }, + { + "epoch": 0.06434963300599926, + "grad_norm": 0.09538089483976364, + "learning_rate": 1.4519436997319034e-05, + "loss": 0.6096773147583008, + "step": 240 + }, + { + "epoch": 0.06971210242316586, + "grad_norm": 0.07478228211402893, + "learning_rate": 1.447922252010724e-05, + "loss": 0.6299086093902588, + "step": 260 + }, + { + "epoch": 0.07507457184033248, + "grad_norm": 0.1514953374862671, + "learning_rate": 1.4439008042895443e-05, + "loss": 0.5591042518615723, + "step": 280 + }, + { + "epoch": 0.08043704125749908, + "grad_norm": 0.08260886371135712, + "learning_rate": 1.4398793565683647e-05, + "loss": 0.6200376987457276, + "step": 300 + }, + { + "epoch": 0.08579951067466568, + "grad_norm": 0.17698714137077332, + "learning_rate": 1.435857908847185e-05, + "loss": 0.6023219585418701, + "step": 320 + }, + { + "epoch": 0.0911619800918323, + "grad_norm": 0.06104859337210655, + "learning_rate": 1.4318364611260054e-05, + "loss": 0.6181454658508301, + "step": 340 + }, + { + "epoch": 0.0965244495089989, + "grad_norm": 0.04990549385547638, + "learning_rate": 1.4278150134048258e-05, + "loss": 0.5593632698059082, + "step": 360 + }, + { + "epoch": 0.1018869189261655, + "grad_norm": 0.09426380693912506, + "learning_rate": 1.4237935656836461e-05, + "loss": 0.5790591716766358, + "step": 380 + }, + { + "epoch": 0.1072493883433321, + "grad_norm": 0.08783263713121414, + "learning_rate": 1.4197721179624665e-05, + "loss": 0.585063886642456, + "step": 400 + }, + { + "epoch": 0.11261185776049872, + "grad_norm": 0.06869607418775558, + "learning_rate": 1.4157506702412869e-05, + "loss": 0.5638764381408692, + "step": 420 + }, + { + "epoch": 0.11797432717766532, + "grad_norm": 0.10537438839673996, + "learning_rate": 1.4117292225201072e-05, + "loss": 0.6060166835784913, + "step": 440 + }, + { + "epoch": 0.12333679659483192, + "grad_norm": 0.09851580113172531, + "learning_rate": 1.4077077747989278e-05, + "loss": 0.5605969905853272, + "step": 460 + }, + { + "epoch": 0.12869926601199852, + "grad_norm": 0.11954096704721451, + "learning_rate": 1.4036863270777482e-05, + "loss": 0.5549856662750244, + "step": 480 + }, + { + "epoch": 0.13406173542916514, + "grad_norm": 0.13259431719779968, + "learning_rate": 1.3996648793565685e-05, + "loss": 0.5893547534942627, + "step": 500 + }, + { + "epoch": 0.13942420484633172, + "grad_norm": 0.11842650175094604, + "learning_rate": 1.3956434316353889e-05, + "loss": 0.6237683773040772, + "step": 520 + }, + { + "epoch": 0.14478667426349834, + "grad_norm": 0.1204022690653801, + "learning_rate": 1.3916219839142093e-05, + "loss": 0.572803258895874, + "step": 540 + }, + { + "epoch": 0.15014914368066495, + "grad_norm": 0.1345946341753006, + "learning_rate": 1.3876005361930296e-05, + "loss": 0.5632933139801025, + "step": 560 + }, + { + "epoch": 0.15551161309783154, + "grad_norm": 0.11733393371105194, + "learning_rate": 1.38357908847185e-05, + "loss": 0.6197309494018555, + "step": 580 + }, + { + "epoch": 0.16087408251499816, + "grad_norm": 0.0731734186410904, + "learning_rate": 1.3795576407506704e-05, + "loss": 0.5823808670043945, + "step": 600 + }, + { + "epoch": 0.16623655193216477, + "grad_norm": 0.09452618658542633, + "learning_rate": 1.3755361930294907e-05, + "loss": 0.5599356651306152, + "step": 620 + }, + { + "epoch": 0.17159902134933136, + "grad_norm": 0.09183815121650696, + "learning_rate": 1.3715147453083111e-05, + "loss": 0.5465828895568847, + "step": 640 + }, + { + "epoch": 0.17696149076649798, + "grad_norm": 0.0953364372253418, + "learning_rate": 1.3674932975871315e-05, + "loss": 0.5516108989715576, + "step": 660 + }, + { + "epoch": 0.1823239601836646, + "grad_norm": 0.11190114170312881, + "learning_rate": 1.3634718498659519e-05, + "loss": 0.5717048645019531, + "step": 680 + }, + { + "epoch": 0.18768642960083118, + "grad_norm": 0.11502158641815186, + "learning_rate": 1.3594504021447722e-05, + "loss": 0.528355598449707, + "step": 700 + }, + { + "epoch": 0.1930488990179978, + "grad_norm": 0.12480133026838303, + "learning_rate": 1.3554289544235926e-05, + "loss": 0.5860391616821289, + "step": 720 + }, + { + "epoch": 0.19841136843516438, + "grad_norm": 0.14408785104751587, + "learning_rate": 1.351407506702413e-05, + "loss": 0.5422697544097901, + "step": 740 + }, + { + "epoch": 0.203773837852331, + "grad_norm": 0.12405668199062347, + "learning_rate": 1.3473860589812333e-05, + "loss": 0.5876667499542236, + "step": 760 + }, + { + "epoch": 0.2091363072694976, + "grad_norm": 0.12171291559934616, + "learning_rate": 1.3433646112600537e-05, + "loss": 0.563751220703125, + "step": 780 + }, + { + "epoch": 0.2144987766866642, + "grad_norm": 0.10827518254518509, + "learning_rate": 1.339343163538874e-05, + "loss": 0.5700247764587403, + "step": 800 + } + ], + "logging_steps": 20, + "max_steps": 7460, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 200, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.803862124388966e+16, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/checkpoint-800/training_args.bin b/checkpoint-800/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/checkpoint-800/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684 +size 11421991 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,29 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|endoftext|>", + "errors": "replace", + "extra_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "is_local": false, + "model_max_length": 131072, + "pad_token": "<|endoftext|>", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null +} diff --git a/training_args.bin b/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c --- /dev/null +++ b/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4 +size 5201